[llvm] [AMDGPU] Promote allocas used by phi/select (PR #221870)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 17:46:19 PDT 2026
https://github.com/ruiling updated https://github.com/llvm/llvm-project/pull/221870
>From 56907331af61437526f9bbeeaf7acec4b4461ec7 Mon Sep 17 00:00:00 2001
From: Ruiling Song <ruiling.song at amd.com>
Date: Mon, 7 Sep 2026 11:30:42 +0800
Subject: [PATCH 1/9] [AMDGPU] Promote allocas used by phi/select
Standard optimization SROA likes to unfold gep(phi) into phi(gep), see
the related PR #83087 and #80983. Such kind of form would trigger alloca
split in next SROA run. This is bad for AMDGPU as it turns dynamically
indexed alloca into smaller pieces and using PHI to access them.
To aggressively promote this, we can check for compatible allocas that
are being PHI'ed and promote them as one combined vector with each old
alloca being part of the combined vector.
Memory intrinsics operating on them are rejected for now.
Co-Authored-By: Claude
---
.../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 369 +++++++++++++++---
.../CodeGen/AMDGPU/frame-index-elimination.ll | 4 +-
llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.ll | 4 +-
.../memory-legalizer-store-infinite-loop.ll | 2 +-
.../CodeGen/AMDGPU/promote-alloca-lifetime.ll | 2 +-
.../CodeGen/AMDGPU/promote-alloca-scoring.ll | 2 +-
.../AMDGPU/promote-alloca-to-lds-select.ll | 65 +--
.../AMDGPU/promote-alloca-vector-phi.ll | 236 +++++++++++
.../AMDGPU/required-export-priority.ll | 6 +-
.../AMDGPU/sgpr-scavenge-fi-stack-id.ll | 2 +-
10 files changed, 566 insertions(+), 126 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index 249d993ce81e2..aea62360d9536 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -95,6 +95,10 @@ struct GEPToVectorIndex {
ConstantInt *VarShift = nullptr; // defaults to 0
ConstantInt *ConstIndex = nullptr; // defaults to 0
Value *Full = nullptr;
+ // The root pointer this GEP chain is based on: either a group member alloca
+ // (constant lane) or a pointer phi/select of members (dynamic lane). The
+ // final lane index is (this GEP's within-object offset) + lane(BasePtr).
+ Value *BasePtr = nullptr;
};
struct MemTransferInfo {
@@ -103,16 +107,42 @@ struct MemTransferInfo {
};
// Analysis for planning the different strategies of alloca promotion.
+//
+// An AllocaAnalysis represents a *group* of one or more allocas that must be
+// promoted together because their pointers are merged by a phi or select. The
+// common case is a singleton group (Members == {Alloca}). When promoting to
+// vector, the whole group becomes a single combined vector value; each member
+// occupies a contiguous lane range starting at Vector.BaseLane[member].
struct AllocaAnalysis {
- AllocaInst *Alloca = nullptr;
+ AllocaInst *Alloca = nullptr; // Primary member (lane 0).
DenseSet<Value *> Pointers;
SmallVector<Use *> Uses;
unsigned Score = 0;
- bool HaveSelectOrPHI = false;
+ // True if some phi/select merges this alloca's pointer with null or a
+ // non-alloca object. Such merges are still fine for LDS promotion, but they
+ // disable vector grouping (we can't assign the other operand a lane).
+ bool HaveUnpromotableMerge = false;
+
+ // All member allocas of this group, in lane order (Members[0] == Alloca).
+ SmallVector<AllocaInst *> Members;
+ // Other allocas this group is directly linked to via a phi/select of
+ // pointers. Used to form groups; empty once grouping is complete.
+ SmallVector<AllocaInst *> Links;
+
struct {
FixedVectorType *Ty = nullptr;
+ // Lane offset of each member alloca within the combined vector.
+ DenseMap<AllocaInst *, unsigned> BaseLane;
+ // Cache of computed lane-index values for pointer phis/selects, so cyclic
+ // pointer phis terminate and are only materialized once.
+ DenseMap<Value *, Value *> IndexCache;
SmallVector<Instruction *> Worklist;
SmallVector<Instruction *> UsersToRemove;
+ // Pointer phis/selects that merge group members. Replaced by parallel lane
+ // -index phis/selects and removed (RAUW poison) after promotion. A
+ // SetVector because a merge is a use of multiple members and would
+ // otherwise be recorded once per member.
+ SmallSetVector<Instruction *, 4> PtrMerges;
MapVector<GetElementPtrInst *, GEPToVectorIndex> GEPVectorIdx;
MapVector<MemTransferInst *, MemTransferInfo> TransferInfo;
} Vector;
@@ -121,7 +151,11 @@ struct AllocaAnalysis {
SmallVector<User *> Worklist;
} LDS;
- explicit AllocaAnalysis(AllocaInst *Alloca) : Alloca(Alloca) {}
+ explicit AllocaAnalysis(AllocaInst *Alloca) : Alloca(Alloca) {
+ Members.push_back(Alloca);
+ }
+
+ bool isGroup() const { return Members.size() > 1; }
};
// Shared implementation which can do both promotion to vector and to LDS.
@@ -280,6 +314,51 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
};
SmallVector<Instruction *, 4> WorkList({AA.Alloca});
+
+ // Classify the "other" operand of a pointer phi/select (i.e. an operand not
+ // derived from Cur). Returns false if the merge makes the alloca entirely
+ // unpromotable (merged with an unknown object). Otherwise records links to
+ // any other allocas (for grouping).
+ //
+ // The operand may itself be a chain of phis/selects (e.g. a phi of a phi), so
+ // we look through them to find the set of underlying root allocas. Visited
+ // guards against cyclic pointer phis.
+ SmallPtrSet<Value *, 8> Visited;
+ std::function<bool(Value *)> ClassifyMergeOperand = [&](Value *Other) -> bool {
+ Other = Other->stripPointerCasts();
+ if (AA.Pointers.contains(Other))
+ return true; // Derived from the same pointer.
+ if (!Visited.insert(Other).second)
+ return true; // Already classified (or on a phi cycle).
+
+ if (isa<ConstantPointerNull, ConstantAggregateZero>(Other)) {
+ // Fine for LDS (null stays null), but we can't give it a vector lane.
+ AA.HaveUnpromotableMerge = true;
+ return true;
+ }
+
+ Value *Obj = getUnderlyingObject(Other);
+ // getUnderlyingObject looks through GEPs/casts but not phi/select; recurse
+ // through those to reach the root allocas.
+ if (auto *Phi = dyn_cast<PHINode>(Obj)) {
+ for (Value *In : Phi->incoming_values())
+ if (!ClassifyMergeOperand(In))
+ return false;
+ return true;
+ }
+ if (auto *SI = dyn_cast<SelectInst>(Obj)) {
+ return ClassifyMergeOperand(SI->getTrueValue()) &&
+ ClassifyMergeOperand(SI->getFalseValue());
+ }
+ auto *OtherAI = dyn_cast<AllocaInst>(Obj);
+ if (!OtherAI)
+ return false;
+
+ if (OtherAI != AA.Alloca)
+ AA.Links.push_back(OtherAI);
+ return true;
+ };
+
while (!WorkList.empty()) {
auto *Cur = WorkList.pop_back_val();
if (find(AA.Pointers, Cur) != AA.Pointers.end())
@@ -297,30 +376,23 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
if (isa<GetElementPtrInst>(U.getUser())) {
WorkList.push_back(Inst);
} else if (auto *SI = dyn_cast<SelectInst>(Inst)) {
- // Only promote a select if we know that the other select operand is
- // from another pointer that will also be promoted.
- if (!binaryOpIsDerivedFromSameAlloca(AA.Alloca, Cur, SI, 1, 2))
- return RejectUser(Inst, "select from mixed objects");
+ // A select may merge this alloca's pointer with another promotable
+ // alloca (recorded as a link for grouping), with the same alloca, or
+ // with null. Anything else is rejected.
+ if (!ClassifyMergeOperand(SI->getTrueValue()) ||
+ !ClassifyMergeOperand(SI->getFalseValue()))
+ return RejectUser(Inst, "select from incompatible alloca");
WorkList.push_back(Inst);
- AA.HaveSelectOrPHI = true;
} else if (auto *Phi = dyn_cast<PHINode>(Inst)) {
- // Repeat for phis.
-
- // TODO: Handle more complex cases. We should be able to replace loops
- // over arrays.
- switch (Phi->getNumIncomingValues()) {
- case 1:
- break;
- case 2:
- if (!binaryOpIsDerivedFromSameAlloca(AA.Alloca, Cur, Phi, 0, 1))
- return RejectUser(Inst, "phi from mixed objects");
- break;
- default:
- return RejectUser(Inst, "phi with too many operands");
- }
+ // Repeat for phis. Every incoming value must be derived from a
+ // promotable alloca (this one or a linked one).
+ bool AllOk = true;
+ for (Value *Incoming : Phi->incoming_values())
+ AllOk &= ClassifyMergeOperand(Incoming);
+ if (!AllOk)
+ return RejectUser(Inst, "phi with incompatible alloca");
WorkList.push_back(Inst);
- AA.HaveSelectOrPHI = true;
}
}
}
@@ -376,28 +448,87 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
VGPRBudgetRatio;
std::vector<AllocaAnalysis> Allocas;
- for (Instruction &I : F.getEntryBlock()) {
- if (AllocaInst *AI = dyn_cast<AllocaInst>(&I)) {
+ {
+ // Collect uses for every candidate alloca independently. This also
+ // records phi/select links between distinct allocas (see collectAllocaUses).
+ std::vector<AllocaAnalysis> Raw;
+ DenseMap<AllocaInst *, int> Index; // -> position in Raw, or -1 if rejected.
+ for (Instruction &I : F.getEntryBlock()) {
+ auto *AI = dyn_cast<AllocaInst>(&I);
// Array allocations are probably not worth handling, since an allocation
// of the array type is the canonical form.
- if (!AI->isStaticAlloca() || AI->isArrayAllocation())
+ if (!AI || !AI->isStaticAlloca() || AI->isArrayAllocation())
continue;
-
LLVM_DEBUG(dbgs() << "Analyzing: " << *AI << '\n');
-
AllocaAnalysis AA{AI};
- if (collectAllocaUses(AA)) {
- analyzePromoteToVector(AA);
- if (PromoteToLDS)
- analyzePromoteToLDS(AA);
- if (AA.Vector.Ty || AA.LDS.Enable) {
- scoreAlloca(AA);
- Allocas.push_back(std::move(AA));
+ if (!collectAllocaUses(AA)) {
+ Index[AI] = -1;
+ continue;
+ }
+ Index[AI] = Raw.size();
+ Raw.push_back(std::move(AA));
+ }
+
+ for (int I = 0, E = Raw.size(); I != E; ++I) {
+ bool Promotable = true;
+ for (AllocaInst *Linked : Raw[I].Links) {
+ // If this alloca is not promotable, the whole group is not promotable
+ // as well.
+ if (Index[Linked] == -1) {
+ // The whole group can not be promoted
+ Promotable = false;
+ break;
+ }
+ }
+
+ if (Promotable && !Raw[I].Uses.empty()) {
+ for (AllocaInst *Linked : Raw[I].Links) {
+ // We have an linked alloca which appears earlier, this is not leader.
+ //
+ assert(Index[Linked] > I && "This should be leader alloca\n");
+ // Pull from the other member alloca into the current alloca, which
+ // is the leader.
+ AllocaAnalysis &LeaderAA = Raw[I];
+ AllocaAnalysis &MemberAA = Raw[Index[Linked]];
+
+ // Merge the uses into the leader alloca.
+ append_range(LeaderAA.Uses, MemberAA.Uses);
+ // Clear member uses so that it will be skipped later.
+ MemberAA.Uses.clear();
+
+ LeaderAA.Pointers.insert_range(MemberAA.Pointers);
+ LeaderAA.Members.push_back(MemberAA.Alloca);
+ LeaderAA.HaveUnpromotableMerge |= MemberAA.HaveUnpromotableMerge;
}
+ // We only need to process leader alloca for later steps.
+ Allocas.push_back(Raw[I]);
}
}
}
+ // Allocas that are merged by phi/select would have duplicated Use entries,
+ // remove them while preserving order.
+ for (AllocaAnalysis &AA : Allocas) {
+ if (!AA.isGroup())
+ continue;
+ SmallPtrSet<Use *, 16> Seen;
+ llvm::erase_if(AA.Uses, [&](Use *U) { return !Seen.insert(U).second; });
+ }
+
+ for (AllocaAnalysis &AA : Allocas) {
+ analyzePromoteToVector(AA);
+ // LDS promotion is not group-aware; only attempt it for singletons.
+ if (PromoteToLDS && !AA.isGroup())
+ analyzePromoteToLDS(AA);
+ }
+
+ // Drop non-promotable alloca.
+ llvm::erase_if(Allocas, [](const AllocaAnalysis &AA) {
+ return !AA.Vector.Ty && !AA.LDS.Enable;
+ });
+ for (AllocaAnalysis &AA : Allocas)
+ scoreAlloca(AA);
+
stable_sort(Allocas,
[](const auto &A, const auto &B) { return A.Score > B.Score; });
@@ -413,10 +544,13 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
SetVector<IntrinsicInst *> DeferredIntrs;
for (AllocaAnalysis &AA : Allocas) {
if (AA.Vector.Ty) {
- std::optional<TypeSize> Size = AA.Alloca->getAllocationSize(DL);
- assert(Size); // Expected to succeed on non-array alloca.
- const unsigned AllocaCost = Size->getFixedValue() * 8;
- // First, check if we have enough budget to vectorize this alloca.
+ unsigned AllocaCost = 0;
+ for (AllocaInst *Member : AA.Members) {
+ std::optional<TypeSize> Size = Member->getAllocationSize(DL);
+ assert(Size); // Expected to succeed on non-array alloca.
+ AllocaCost += Size->getFixedValue() * 8;
+ }
+ // First, check if we have enough budget to vectorize this group.
if (AllocaCost <= VectorizationBudget) {
promoteAllocaToVector(AA);
Changed = true;
@@ -464,17 +598,56 @@ static bool isSupportedMemset(MemSetInst *I, AllocaInst *AI,
}
static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
- IRBuilder<> B(Ptr->getContext());
-
+ LLVMContext &Ctx = Ptr->getContext();
Ptr = Ptr->stripPointerCasts();
- if (Ptr == AA.Alloca)
- return B.getInt32(0);
+
+ // Already computed (also breaks cycles for pointer phis).
+ auto Cached = AA.Vector.IndexCache.find(Ptr);
+ if (Cached != AA.Vector.IndexCache.end())
+ return Cached->second;
+
+ const auto BaseLaneConst = [&](AllocaInst *Member) -> Value * {
+ return ConstantInt::get(Type::getInt32Ty(Ctx),
+ AA.Vector.BaseLane.lookup(Member));
+ };
+
+ // A pointer that is directly one of the member allocas indexes lane
+ // BaseLane[member].
+ if (auto *AI = dyn_cast<AllocaInst>(Ptr)) {
+ Value *Idx = BaseLaneConst(AI);
+ AA.Vector.IndexCache[Ptr] = Idx;
+ return Idx;
+ }
+
+ // Pointer phi: build a parallel phi of lane indices. Insert and cache it
+ // before recursing so self-referential phis terminate.
+ if (auto *Phi = dyn_cast<PHINode>(Ptr)) {
+ IRBuilder<> B(Phi);
+ PHINode *IdxPhi =
+ B.CreatePHI(B.getInt32Ty(), Phi->getNumIncomingValues(), "promotealloca.idx");
+ AA.Vector.IndexCache[Ptr] = IdxPhi;
+ for (unsigned I = 0, E = Phi->getNumIncomingValues(); I != E; ++I)
+ IdxPhi->addIncoming(calculateVectorIndex(Phi->getIncomingValue(I), AA),
+ Phi->getIncomingBlock(I));
+ return IdxPhi;
+ }
+
+ // Pointer select: select between the two operand lane indices.
+ if (auto *SI = dyn_cast<SelectInst>(Ptr)) {
+ Value *T = calculateVectorIndex(SI->getTrueValue(), AA);
+ Value *F = calculateVectorIndex(SI->getFalseValue(), AA);
+ IRBuilder<> B(SI);
+ Value *Idx = B.CreateSelect(SI->getCondition(), T, F, "promotealloca.idx");
+ AA.Vector.IndexCache[Ptr] = Idx;
+ return Idx;
+ }
auto *GEP = cast<GetElementPtrInst>(Ptr);
auto I = AA.Vector.GEPVectorIdx.find(GEP);
assert(I != AA.Vector.GEPVectorIdx.end() && "Must have entry for GEP!");
if (!I->second.Full) {
+ IRBuilder<> B(Ctx);
Value *Result = nullptr;
B.SetInsertPoint(GEP);
@@ -499,14 +672,24 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
if (!Result)
Result = B.getInt32(0);
+ // Offset by the lane of the root pointer this GEP is based on. For a member
+ // alloca this is its constant base lane; for a pointer phi/select it is the
+ // parallel lane-index value.
+ if (I->second.BasePtr != AA.Alloca || AA.isGroup()) {
+ Value *BaseIdx = calculateVectorIndex(I->second.BasePtr, AA);
+ if (auto *C = dyn_cast<ConstantInt>(BaseIdx); !C || !C->isZero())
+ Result = B.CreateAdd(Result, BaseIdx);
+ }
+
I->second.Full = Result;
}
+ AA.Vector.IndexCache[Ptr] = I->second.Full;
return I->second.Full;
}
static std::optional<GEPToVectorIndex>
-computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
+computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaAnalysis &AA,
Type *VecElemTy, const DataLayout &DL) {
// TODO: Extracting a "multiple of X" from a GEP might be a useful generic
// helper.
@@ -541,7 +724,9 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
CurPtr = CurGEP->getPointerOperand();
}
- assert(CurPtr == Alloca && "GEP not based on alloca");
+ // This should either points to the alloca or known phi/select.
+ assert(isa<AllocaInst>(CurPtr) ||
+ (isa<PHINode, SelectInst>(CurPtr) /*&& AA.Pointers.contains(CurPtr)*/));
int64_t VecElemSize = DL.getTypeAllocSize(VecElemTy);
if (VarOffsets.size() > 1)
@@ -555,6 +740,7 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
APInt IndexQuot = ConstOffset.sdiv(VecElemSize);
GEPToVectorIndex Result;
+ Result.BasePtr = CurPtr;
if (!ConstOffset.isZero())
Result.ConstIndex = ConstantInt::get(Ctx, IndexQuot.sextOrTrunc(BW));
@@ -903,11 +1089,6 @@ static BasicBlock::iterator skipToNonAllocaInsertPt(BasicBlock &BB,
FixedVectorType *
AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
- if (DisablePromoteAllocaToVector) {
- LLVM_DEBUG(dbgs() << " Promote alloca to vectors is disabled\n");
- return nullptr;
- }
-
auto *VectorTy = dyn_cast<FixedVectorType>(AllocaTy);
if (auto *ArrayTy = dyn_cast<ArrayType>(AllocaTy)) {
uint64_t NumElems = 1;
@@ -966,15 +1147,58 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
}
void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
- if (AA.HaveSelectOrPHI) {
- LLVM_DEBUG(dbgs() << " Cannot convert to vector due to select or phi\n");
+ // A phi/select that merges with null or a non-alloca object can't be given a
+ // vector lane, so the whole group is unpromotable to vector.
+ if (AA.HaveUnpromotableMerge) {
+ LLVM_DEBUG(dbgs() << " Cannot convert to vector due to merge with "
+ "null/unknown object\n");
+ return;
+ }
+
+ if (DisablePromoteAllocaToVector) {
+ LLVM_DEBUG(dbgs() << " Promote alloca to vectors is disabled\n");
return;
}
- Type *AllocaTy = AA.Alloca->getAllocatedType();
- AA.Vector.Ty = getVectorTypeForAlloca(AllocaTy);
- if (!AA.Vector.Ty)
+ // Find a valid common element type and assign each member a contiguous lane
+ // range in the combined vector.
+ Type *CombinedEltTy = nullptr;
+ unsigned TotalElems = 0;
+ for (AllocaInst *Member : AA.Members) {
+ Type *MemberTy = Member->getAllocatedType();
+ unsigned ElemCnt;
+ if (MemberTy->isSingleValueType() && !MemberTy->isVectorTy()) {
+ if (CombinedEltTy && CombinedEltTy != MemberTy)
+ return;
+
+ CombinedEltTy = MemberTy;
+ ElemCnt = 1;
+ } else {
+ MemberTy = getVectorTypeForAlloca(MemberTy);
+ // Failed to find a proper vector type or found a different element type,
+ // abort promotion.
+ if (!MemberTy ||
+ (CombinedEltTy && CombinedEltTy != MemberTy->getScalarType()))
+ return;
+ CombinedEltTy = MemberTy->getScalarType();
+ ElemCnt = cast<FixedVectorType>(MemberTy)->getNumElements();
+ }
+
+
+ AA.Vector.BaseLane[Member] = TotalElems;
+ TotalElems += ElemCnt;
+ }
+
+ AA.Vector.Ty = FixedVectorType::get(CombinedEltTy, TotalElems);
+ // Re-check the combined vector against the register-size limit.
+ const unsigned MaxElements =
+ (MaxVectorRegs * 32) / DL.getTypeSizeInBits(CombinedEltTy);
+ if (TotalElems > MaxElements) {
+ LLVM_DEBUG(dbgs() << " Combined group vector " << *AA.Vector.Ty
+ << " exceeds the register budget\n");
+ AA.Vector.Ty = nullptr;
return;
+ }
const auto RejectUser = [&](Instruction *Inst, Twine Msg) {
LLVM_DEBUG(dbgs() << " Cannot promote alloca to vector: " << Msg << "\n"
@@ -988,6 +1212,13 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
for (auto *U : AA.Uses) {
Instruction *Inst = cast<Instruction>(U->getUser());
+ // Pointer phis/selects merging group members are handled by rewriting them
+ // to lane-index phis/selects during promotion.
+ if (isa<PHINode, SelectInst>(Inst) && Inst->getType()->isPointerTy()) {
+ AA.Vector.PtrMerges.insert(Inst);
+ continue;
+ }
+
if (Value *Ptr = getLoadStorePointerOperand(Inst)) {
assert(!isa<StoreInst>(Inst) ||
U->getOperandNo() == StoreInst::getPointerOperandIndex());
@@ -1005,8 +1236,8 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
Ptr = Ptr->stripPointerCasts();
- // Alloca already accessed as vector.
- if (Ptr == AA.Alloca &&
+ // Alloca already accessed as vector for single alloca group case.
+ if (!AA.isGroup() && Ptr == AA.Alloca &&
DL.getTypeStoreSize(AA.Alloca->getAllocatedType()) ==
DL.getTypeStoreSize(AccessTy)) {
AA.Vector.Worklist.push_back(Inst);
@@ -1023,7 +1254,7 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
if (auto *GEP = dyn_cast<GetElementPtrInst>(Inst)) {
// If we can't compute a vector index from this GEP, then we can't
// promote this alloca to vector.
- auto Index = computeGEPToVectorIndex(GEP, AA.Alloca, VecEltTy, DL);
+ auto Index = computeGEPToVectorIndex(GEP, AA, VecEltTy, DL);
if (!Index)
return RejectUser(Inst, "cannot compute vector index for GEP");
@@ -1033,12 +1264,14 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
}
if (MemSetInst *MSI = dyn_cast<MemSetInst>(Inst);
- MSI && isSupportedMemset(MSI, AA.Alloca, DL)) {
+ MSI && !AA.isGroup() && isSupportedMemset(MSI, AA.Alloca, DL)) {
AA.Vector.Worklist.push_back(Inst);
continue;
}
if (MemTransferInst *TransferInst = dyn_cast<MemTransferInst>(Inst)) {
+ if (AA.isGroup())
+ return RejectUser(Inst, "mem transfer in grouped alloca not supported");
if (TransferInst->isVolatile())
return RejectUser(Inst, "mem transfer inst is volatile");
@@ -1187,6 +1420,16 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
I->eraseFromParent();
}
+ // The pointer phis/selects merging group members are now dead. Drop them
+ // before the GEPs they reference. RAUW with poison first so chained merges
+ // can be erased in any order.
+ for (Instruction *I : AA.Vector.PtrMerges)
+ I->replaceAllUsesWith(PoisonValue::get(I->getType()));
+ for (Instruction *I : AA.Vector.PtrMerges) {
+ assert(I->use_empty());
+ I->eraseFromParent();
+ }
+
// Delete all the users that are known to be removeable.
for (Instruction *I : reverse(AA.Vector.UsersToRemove)) {
I->dropDroppableUses();
@@ -1194,9 +1437,11 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
I->eraseFromParent();
}
- // Alloca should now be dead too.
- assert(AA.Alloca->use_empty());
- AA.Alloca->eraseFromParent();
+ // All member allocas should now be dead too.
+ for (AllocaInst *Member : AA.Members) {
+ assert(Member->use_empty());
+ Member->eraseFromParent();
+ }
}
std::pair<Value *, Value *>
diff --git a/llvm/test/CodeGen/AMDGPU/frame-index-elimination.ll b/llvm/test/CodeGen/AMDGPU/frame-index-elimination.ll
index bf4d3ce25a6ef..6f6a5c41cf11c 100644
--- a/llvm/test/CodeGen/AMDGPU/frame-index-elimination.ll
+++ b/llvm/test/CodeGen/AMDGPU/frame-index-elimination.ll
@@ -1,8 +1,8 @@
; RUN: llc -mtriple=amdgpu7.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds < %s | FileCheck -enable-var-scope -check-prefixes=GCN,CI,MUBUF %s
; RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds < %s | FileCheck -enable-var-scope -check-prefixes=GCN,GFX9,GFX9-MUBUF,MUBUF %s
; RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds -mattr=+enable-flat-scratch < %s | FileCheck -enable-var-scope -check-prefixes=GCN,GFX9,GFX9-FLATSCR %s
-; RUN: llc -mtriple=amdgpu11.00-amd-amdhsa -mattr=+real-true16 < %s | FileCheck --check-prefixes=GFX11-TRUE16 %s
-; RUN: llc -mtriple=amdgpu11.00-amd-amdhsa -mattr=-real-true16 < %s | FileCheck --check-prefixes=GFX11-FAKE16 %s
+; RUN: llc -mtriple=amdgpu11.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds -mattr=+real-true16 < %s | FileCheck --check-prefixes=GFX11-TRUE16 %s
+; RUN: llc -mtriple=amdgpu11.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds -mattr=-real-true16 < %s | FileCheck --check-prefixes=GFX11-FAKE16 %s
; Test that non-entry function frame indices are expanded properly to
; give an index relative to the scratch wave offset register
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.ll
index 767aa8228fa0a..59f64a4f6975b 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.ll
@@ -639,8 +639,8 @@ define amdgpu_kernel void @test_export_pos_before_param_across_load(i32 %idx) #0
; PREGFX11: {{exp|export}} param0
; PREGFX11: {{exp|export}} param1
define amdgpu_kernel void @test_export_across_store_load(i32 %idx, float %v) #0 {
- %data0 = alloca <4 x float>, align 8, addrspace(5)
- %data1 = alloca <4 x float>, align 8, addrspace(5)
+ %data0 = alloca <64 x float>, align 8, addrspace(5)
+ %data1 = alloca <64 x float>, align 8, addrspace(5)
%cmp = icmp eq i32 %idx, 1
%data = select i1 %cmp, ptr addrspace(5) %data0, ptr addrspace(5) %data1
store float %v, ptr addrspace(5) %data, align 8
diff --git a/llvm/test/CodeGen/AMDGPU/memory-legalizer-store-infinite-loop.ll b/llvm/test/CodeGen/AMDGPU/memory-legalizer-store-infinite-loop.ll
index 2786515f1982e..880f8d3972ca1 100644
--- a/llvm/test/CodeGen/AMDGPU/memory-legalizer-store-infinite-loop.ll
+++ b/llvm/test/CodeGen/AMDGPU/memory-legalizer-store-infinite-loop.ll
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -mtriple=amdgpu7.00--amdhsa < %s | FileCheck -check-prefix=GCN %s
+; RUN: llc -mtriple=amdgpu7.00--amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds < %s | FileCheck -check-prefix=GCN %s
; Effectively, check that the compile finishes; in the case
; of an infinite loop, llc toggles between merging 2 ST4s
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-lifetime.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-lifetime.ll
index 1dd85cafbaa06..0b0cf9a14acbe 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-lifetime.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-lifetime.ll
@@ -1,4 +1,4 @@
-; RUN: opt -S -mtriple=amdgpu-unknown-amdhsa -passes=amdgpu-promote-alloca %s | FileCheck -check-prefix=OPT %s
+; RUN: opt -S -mtriple=amdgpu-unknown-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-vector %s | FileCheck -check-prefix=OPT %s
declare void @llvm.lifetime.start.p5(i64, ptr addrspace(5) nocapture) #0
declare void @llvm.lifetime.end.p5(i64, ptr addrspace(5) nocapture) #0
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll
index 09643a826f003..473fa92547a43 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll
@@ -2,10 +2,10 @@
; REQUIRES: asserts
; CHECK-LABEL: Analyzing: %simpleuser = alloca [4 x i64], align 4, addrspace(5)
+; CHECK-NEXT: Analyzing: %manyusers = alloca [4 x i64], align 4, addrspace(5)
; CHECK-NEXT: Scoring: %simpleuser = alloca [4 x i64], align 4, addrspace(5)
; CHECK-NEXT: [+1]: store i64 42, ptr addrspace(5) %simpleuser, align 8
; CHECK-NEXT: => Final Score:1
-; CHECK-LABEL: Analyzing: %manyusers = alloca [4 x i64], align 4, addrspace(5)
; CHECK-NEXT: Scoring: %manyusers = alloca [4 x i64], align 4, addrspace(5)
; CHECK-NEXT: [+1]: store i64 %v0.add, ptr addrspace(5) %manyusers.1, align 8
; CHECK-NEXT: [+1]: %v0 = load i64, ptr addrspace(5) %manyusers.1, align 8
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
index b38c2d5b20565..e7bb3017bc07b 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
@@ -18,25 +18,9 @@ define amdgpu_kernel void @lds_promoted_alloca_select_invalid_pointer_operand()
define amdgpu_kernel void @lds_promote_alloca_select_two_derived_pointers(i32 %a, i32 %b) #0 {
; CHECK-LABEL: define amdgpu_kernel void @lds_promote_alloca_select_two_derived_pointers(
; CHECK-SAME: i32 [[A:%.*]], i32 [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[TMP1:%.*]] = call noalias nonnull dereferenceable(64) ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
-; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 1
-; CHECK-NEXT: [[TMP3:%.*]] = load i32, ptr addrspace(4) [[TMP2]], align 4, !invariant.load [[META0:![0-9]+]]
-; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 2
-; CHECK-NEXT: [[TMP5:%.*]] = load i32, ptr addrspace(4) [[TMP4]], align 4, !range [[RNG1:![0-9]+]], !invariant.load [[META0]]
-; CHECK-NEXT: [[TMP6:%.*]] = lshr i32 [[TMP3]], 16
-; CHECK-NEXT: [[TMP7:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.x()
-; CHECK-NEXT: [[TMP8:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.y()
-; CHECK-NEXT: [[TMP9:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.z()
-; CHECK-NEXT: [[TMP10:%.*]] = mul nuw nsw i32 [[TMP6]], [[TMP5]]
-; CHECK-NEXT: [[TMP11:%.*]] = mul i32 [[TMP10]], [[TMP7]]
-; CHECK-NEXT: [[TMP12:%.*]] = mul nuw nsw i32 [[TMP8]], [[TMP5]]
-; CHECK-NEXT: [[TMP13:%.*]] = add i32 [[TMP11]], [[TMP12]]
-; CHECK-NEXT: [[TMP14:%.*]] = add i32 [[TMP13]], [[TMP9]]
-; CHECK-NEXT: [[TMP15:%.*]] = getelementptr inbounds [256 x [16 x i32]], ptr addrspace(3) @lds_promote_alloca_select_two_derived_pointers.alloca, i32 0, i32 [[TMP14]]
-; CHECK-NEXT: [[PTR0:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 [[A]]
-; CHECK-NEXT: [[PTR1:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 [[B]]
-; CHECK-NEXT: [[SELECT:%.*]] = select i1 poison, ptr addrspace(3) [[PTR0]], ptr addrspace(3) [[PTR1]]
-; CHECK-NEXT: store i32 0, ptr addrspace(3) [[SELECT]], align 4
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <16 x i32> poison
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX:%.*]] = select i1 poison, i32 [[A]], i32 [[B]]
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <16 x i32> [[ALLOCA]], i32 0, i32 [[PROMOTEALLOCA_IDX]]
; CHECK-NEXT: ret void
;
%alloca = alloca [16 x i32], align 4, addrspace(5)
@@ -73,25 +57,7 @@ define amdgpu_kernel void @lds_promote_alloca_select_two_allocas(i32 %a, i32 %b)
define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointers() #0 {
; CHECK-LABEL: define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointers(
; CHECK-SAME: ) #[[ATTR0]] {
-; CHECK-NEXT: [[TMP1:%.*]] = call noalias nonnull dereferenceable(64) ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
-; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 1
-; CHECK-NEXT: [[TMP3:%.*]] = load i32, ptr addrspace(4) [[TMP2]], align 4, !invariant.load [[META0]]
-; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 2
-; CHECK-NEXT: [[TMP5:%.*]] = load i32, ptr addrspace(4) [[TMP4]], align 4, !range [[RNG1]], !invariant.load [[META0]]
-; CHECK-NEXT: [[TMP6:%.*]] = lshr i32 [[TMP3]], 16
-; CHECK-NEXT: [[TMP7:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.x()
-; CHECK-NEXT: [[TMP8:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.y()
-; CHECK-NEXT: [[TMP9:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.z()
-; CHECK-NEXT: [[TMP10:%.*]] = mul nuw nsw i32 [[TMP6]], [[TMP5]]
-; CHECK-NEXT: [[TMP11:%.*]] = mul i32 [[TMP10]], [[TMP7]]
-; CHECK-NEXT: [[TMP12:%.*]] = mul nuw nsw i32 [[TMP8]], [[TMP5]]
-; CHECK-NEXT: [[TMP13:%.*]] = add i32 [[TMP11]], [[TMP12]]
-; CHECK-NEXT: [[TMP14:%.*]] = add i32 [[TMP13]], [[TMP9]]
-; CHECK-NEXT: [[TMP15:%.*]] = getelementptr inbounds [256 x [16 x i32]], ptr addrspace(3) @lds_promote_alloca_select_two_derived_constant_pointers.alloca, i32 0, i32 [[TMP14]]
-; CHECK-NEXT: [[PTR0:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 1
-; CHECK-NEXT: [[PTR1:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 3
-; CHECK-NEXT: [[SELECT:%.*]] = select i1 poison, ptr addrspace(3) [[PTR0]], ptr addrspace(3) [[PTR1]]
-; CHECK-NEXT: store i32 0, ptr addrspace(3) [[SELECT]], align 4
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <16 x i32> poison
; CHECK-NEXT: ret void
;
%alloca = alloca [16 x i32], align 4, addrspace(5)
@@ -102,19 +68,13 @@ define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointe
ret void
}
-; FIXME: Can be promoted, but we'd have to recursively show that the select
-; operands all point to the same alloca.
-
define amdgpu_kernel void @lds_promoted_alloca_select_input_select(i32 %a, i32 %b, i32 %c, i1 %c1, i1 %c2) #0 {
; CHECK-LABEL: define amdgpu_kernel void @lds_promoted_alloca_select_input_select(
; CHECK-SAME: i32 [[A:%.*]], i32 [[B:%.*]], i32 [[C:%.*]], i1 [[C1:%.*]], i1 [[C2:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[ALLOCA:%.*]] = alloca [16 x i32], align 4, addrspace(5)
-; CHECK-NEXT: [[PTR0:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(5) [[ALLOCA]], i32 0, i32 [[A]]
-; CHECK-NEXT: [[PTR1:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(5) [[ALLOCA]], i32 0, i32 [[B]]
-; CHECK-NEXT: [[PTR2:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(5) [[ALLOCA]], i32 0, i32 [[C]]
-; CHECK-NEXT: [[SELECT0:%.*]] = select i1 [[C1]], ptr addrspace(5) [[PTR0]], ptr addrspace(5) [[PTR1]]
-; CHECK-NEXT: [[SELECT1:%.*]] = select i1 [[C2]], ptr addrspace(5) [[SELECT0]], ptr addrspace(5) [[PTR2]]
-; CHECK-NEXT: store i32 0, ptr addrspace(5) [[SELECT1]], align 4
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <16 x i32> poison
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX:%.*]] = select i1 [[C1]], i32 [[A]], i32 [[B]]
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX1:%.*]] = select i1 [[C2]], i32 [[PROMOTEALLOCA_IDX]], i32 [[C]]
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <16 x i32> [[ALLOCA]], i32 0, i32 [[PROMOTEALLOCA_IDX1]]
; CHECK-NEXT: ret void
;
%alloca = alloca [16 x i32], align 4, addrspace(5)
@@ -173,9 +133,9 @@ define amdgpu_kernel void @select_null_rhs(ptr addrspace(1) nocapture %arg, i32
; CHECK-NEXT: [[BB:.*:]]
; CHECK-NEXT: [[TMP0:%.*]] = call noalias nonnull dereferenceable(64) ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 1
-; CHECK-NEXT: [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4, !invariant.load [[META0]]
+; CHECK-NEXT: [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4, !invariant.load [[META0:![0-9]+]]
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 2
-; CHECK-NEXT: [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG2:![0-9]+]], !invariant.load [[META0]]
+; CHECK-NEXT: [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG1:![0-9]+]], !invariant.load [[META0]]
; CHECK-NEXT: [[TMP5:%.*]] = lshr i32 [[TMP2]], 16
; CHECK-NEXT: [[TMP6:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.x()
; CHECK-NEXT: [[TMP7:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.y()
@@ -213,7 +173,7 @@ define amdgpu_kernel void @select_null_lhs(ptr addrspace(1) nocapture %arg, i32
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 1
; CHECK-NEXT: [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4, !invariant.load [[META0]]
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 2
-; CHECK-NEXT: [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG2]], !invariant.load [[META0]]
+; CHECK-NEXT: [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG1]], !invariant.load [[META0]]
; CHECK-NEXT: [[TMP5:%.*]] = lshr i32 [[TMP2]], 16
; CHECK-NEXT: [[TMP6:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.x()
; CHECK-NEXT: [[TMP7:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.y()
@@ -247,6 +207,5 @@ attributes #0 = { norecurse nounwind "amdgpu-waves-per-eu"="1,1" "amdgpu-flat-wo
attributes #1 = { norecurse nounwind }
;.
; CHECK: [[META0]] = !{}
-; CHECK: [[RNG1]] = !{i32 0, i32 257}
-; CHECK: [[RNG2]] = !{i32 0, i32 1025}
+; CHECK: [[RNG1]] = !{i32 0, i32 1025}
;.
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
new file mode 100644
index 0000000000000..b57874848e497
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
@@ -0,0 +1,236 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu-unknown-amdhsa -passes=amdgpu-promote-alloca-to-vector < %s | FileCheck %s
+
+; Tests for promotion of allocas whose pointers are merged by phis/selects,
+
+define amdgpu_kernel void @phi_two_allocas_basic(i1 %c, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_two_allocas_basic(
+; CHECK-SAME: i1 [[C:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[A:%.*]] = freeze <2 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x i32> [[A]], i32 1, i32 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x i32> [[TMP0]], i32 2, i32 1
+; CHECK-NEXT: br i1 [[C]], label %[[BB1:.*]], label %[[BB2:.*]]
+; CHECK: [[BB1]]:
+; CHECK-NEXT: br label %[[JOIN:.*]]
+; CHECK: [[BB2]]:
+; CHECK-NEXT: br label %[[JOIN]]
+; CHECK: [[JOIN]]:
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX:%.*]] = phi i32 [ 0, %[[BB1]] ], [ 1, %[[BB2]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = extractelement <2 x i32> [[TMP1]], i32 [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %a = alloca i32, addrspace(5)
+ %b = alloca i32, addrspace(5)
+ store i32 1, ptr addrspace(5) %a
+ store i32 2, ptr addrspace(5) %b
+ br i1 %c, label %bb.1, label %bb.2
+bb.1:
+ br label %join
+bb.2:
+ br label %join
+join:
+ %p = phi ptr addrspace(5) [ %a, %bb.1 ], [ %b, %bb.2 ]
+ %l = load i32, ptr addrspace(5) %p
+ store i32 %l, ptr addrspace(1) %out
+ ret void
+}
+
+; GEP uses a (chained) pointer phi: phi-of-phi feeding a gep.
+define amdgpu_kernel void @gep_uses_chained_phi(i1 %c, i1 %d, i32 %v, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @gep_uses_chained_phi(
+; CHECK-SAME: i1 [[C:%.*]], i1 [[D:%.*]], i32 [[V:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[A:%.*]] = freeze <8 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <8 x i32> [[A]], i32 [[V]], i32 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <8 x i32> [[TMP0]], i32 [[V]], i32 4
+; CHECK-NEXT: br i1 [[C]], label %[[BB_1:.*]], label %[[BB_2:.*]]
+; CHECK: [[BB_1]]:
+; CHECK-NEXT: br label %[[JOIN1:.*]]
+; CHECK: [[BB_2]]:
+; CHECK-NEXT: br label %[[JOIN1]]
+; CHECK: [[JOIN1]]:
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX1:%.*]] = phi i32 [ 0, %[[BB_1]] ], [ 4, %[[BB_2]] ]
+; CHECK-NEXT: br i1 [[D]], label %[[BB_3:.*]], label %[[JOIN2:.*]]
+; CHECK: [[BB_3]]:
+; CHECK-NEXT: br label %[[JOIN2]]
+; CHECK: [[JOIN2]]:
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX:%.*]] = phi i32 [ [[PROMOTEALLOCA_IDX1]], %[[JOIN1]] ], [ 0, %[[BB_3]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = add i32 1, [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT: [[TMP3:%.*]] = extractelement <8 x i32> [[TMP1]], i32 [[TMP2]]
+; CHECK-NEXT: store i32 [[TMP3]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %a = alloca [4 x i32], addrspace(5)
+ %b = alloca [4 x i32], addrspace(5)
+ store i32 %v, ptr addrspace(5) %a
+ store i32 %v, ptr addrspace(5) %b
+ br i1 %c, label %bb.1, label %bb.2
+bb.1:
+ br label %join1
+bb.2:
+ br label %join1
+join1:
+ %p1 = phi ptr addrspace(5) [ %a, %bb.1 ], [ %b, %bb.2 ]
+ br i1 %d, label %bb.3, label %join2
+bb.3:
+ br label %join2
+join2:
+ %p2 = phi ptr addrspace(5) [ %p1, %join1 ], [ %a, %bb.3 ]
+ %g = getelementptr i32, ptr addrspace(5) %p2, i32 1
+ %l = load i32, ptr addrspace(5) %g
+ store i32 %l, ptr addrspace(1) %out
+ ret void
+}
+
+; Pointer phi whose incoming values are GEPs (phi uses gep), and a gep uses the
+; phi. Exercises both erase-order directions in a single group.
+define amdgpu_kernel void @merge_uses_gep(i1 %c, i32 %v, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @merge_uses_gep(
+; CHECK-SAME: i1 [[C:%.*]], i32 [[V:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[B:%.*]] = freeze <8 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <8 x i32> [[B]], i32 [[V]], i32 2
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <8 x i32> [[TMP0]], i32 [[V]], i32 7
+; CHECK-NEXT: br i1 [[C]], label %[[BB1:.*]], label %[[BB2:.*]]
+; CHECK: [[BB1]]:
+; CHECK-NEXT: br label %[[JOIN:.*]]
+; CHECK: [[BB2]]:
+; CHECK-NEXT: br label %[[JOIN]]
+; CHECK: [[JOIN]]:
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX:%.*]] = phi i32 [ 2, %[[BB1]] ], [ 7, %[[BB2]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = add i32 1, [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT: [[TMP3:%.*]] = extractelement <8 x i32> [[TMP1]], i32 [[TMP2]]
+; CHECK-NEXT: store i32 [[TMP3]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %a = alloca [4 x i32], addrspace(5)
+ %b = alloca [4 x i32], addrspace(5)
+ %ga = getelementptr i32, ptr addrspace(5) %a, i32 2
+ %gb = getelementptr i32, ptr addrspace(5) %b, i32 3
+ store i32 %v, ptr addrspace(5) %ga
+ store i32 %v, ptr addrspace(5) %gb
+ br i1 %c, label %bb.1, label %bb.2
+bb.1:
+ br label %join
+bb.2:
+ br label %join
+join:
+ %p = phi ptr addrspace(5) [ %ga, %bb.1 ], [ %gb, %bb.2 ]
+ %g2 = getelementptr i32, ptr addrspace(5) %p, i32 1
+ %l = load i32, ptr addrspace(5) %g2
+ store i32 %l, ptr addrspace(1) %out
+ ret void
+}
+
+; Chained selects: select of a select.
+define amdgpu_kernel void @select_chain(i1 %c, i1 %d, i32 %v, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @select_chain(
+; CHECK-SAME: i1 [[C:%.*]], i1 [[D:%.*]], i32 [[V:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[B:%.*]] = freeze <8 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <8 x i32> [[B]], i32 [[V]], i32 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <8 x i32> [[TMP0]], i32 [[V]], i32 4
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX:%.*]] = select i1 [[C]], i32 0, i32 4
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX1:%.*]] = select i1 [[D]], i32 [[PROMOTEALLOCA_IDX]], i32 0
+; CHECK-NEXT: [[TMP2:%.*]] = add i32 1, [[PROMOTEALLOCA_IDX1]]
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <8 x i32> [[TMP1]], i32 [[V]], i32 [[TMP2]]
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <8 x i32> [[TMP3]], i32 [[PROMOTEALLOCA_IDX1]]
+; CHECK-NEXT: store i32 [[TMP4]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %a = alloca [4 x i32], addrspace(5)
+ %b = alloca [4 x i32], addrspace(5)
+ store i32 %v, ptr addrspace(5) %a
+ store i32 %v, ptr addrspace(5) %b
+ %s1 = select i1 %c, ptr addrspace(5) %a, ptr addrspace(5) %b
+ %s2 = select i1 %d, ptr addrspace(5) %s1, ptr addrspace(5) %a
+ %g = getelementptr i32, ptr addrspace(5) %s2, i32 1
+ store i32 %v, ptr addrspace(5) %g
+ %l = load i32, ptr addrspace(5) %s2
+ store i32 %l, ptr addrspace(1) %out
+ ret void
+}
+
+; Self-referential pointer phi across a loop back-edge.
+define amdgpu_kernel void @loop_backedge_phi(i32 %n, i32 %v, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @loop_backedge_phi(
+; CHECK-SAME: i32 [[N:%.*]], i32 [[V:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[A:%.*]] = freeze <8 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <8 x i32> [[A]], i32 [[V]], i32 0
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[I:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[TMP2:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = extractelement <8 x i32> [[TMP0]], i32 [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT: store i32 [[TMP1]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: [[TMP2]] = add i32 1, [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT: [[I_NEXT]] = add i32 [[I]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %a = alloca [8 x i32], addrspace(5)
+ store i32 %v, ptr addrspace(5) %a
+ br label %loop
+loop:
+ %i = phi i32 [ 0, %entry ], [ %i.next, %loop ]
+ %p = phi ptr addrspace(5) [ %a, %entry ], [ %p.next, %loop ]
+ %l = load i32, ptr addrspace(5) %p
+ store i32 %l, ptr addrspace(1) %out
+ %p.next = getelementptr i32, ptr addrspace(5) %p, i32 1
+ %i.next = add i32 %i, 1
+ %cmp = icmp slt i32 %i.next, %n
+ br i1 %cmp, label %loop, label %exit
+exit:
+ ret void
+}
+
+; Group members of different lengths combine into one wider vector:
+; [4 x i32] + [2 x i32] -> <6 x i32>, with each member at a contiguous lane range.
+define amdgpu_kernel void @phi_diff_len(i1 %c, i32 %v, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_diff_len(
+; CHECK-SAME: i1 [[C:%.*]], i32 [[V:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[B:%.*]] = freeze <6 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <6 x i32> [[B]], i32 [[V]], i32 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <6 x i32> [[TMP0]], i32 [[V]], i32 4
+; CHECK-NEXT: br i1 [[C]], label %[[BB1:.*]], label %[[BB2:.*]]
+; CHECK: [[BB1]]:
+; CHECK-NEXT: br label %[[JOIN:.*]]
+; CHECK: [[BB2]]:
+; CHECK-NEXT: br label %[[JOIN]]
+; CHECK: [[JOIN]]:
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX:%.*]] = phi i32 [ 0, %[[BB1]] ], [ 4, %[[BB2]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = add i32 1, [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <6 x i32> [[TMP1]], i32 [[V]], i32 [[TMP2]]
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <6 x i32> [[TMP3]], i32 [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT: store i32 [[TMP4]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %a = alloca [4 x i32], addrspace(5)
+ %b = alloca [2 x i32], addrspace(5)
+ store i32 %v, ptr addrspace(5) %a
+ store i32 %v, ptr addrspace(5) %b
+ br i1 %c, label %bb.1, label %bb.2
+bb.1:
+ br label %join
+bb.2:
+ br label %join
+join:
+ %p = phi ptr addrspace(5) [ %a, %bb.1 ], [ %b, %bb.2 ]
+ %g = getelementptr i32, ptr addrspace(5) %p, i32 1
+ store i32 %v, ptr addrspace(5) %g
+ %l = load i32, ptr addrspace(5) %p
+ store i32 %l, ptr addrspace(1) %out
+ ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/required-export-priority.ll b/llvm/test/CodeGen/AMDGPU/required-export-priority.ll
index b6e2e75c9d28c..4df215a3f6224 100644
--- a/llvm/test/CodeGen/AMDGPU/required-export-priority.ll
+++ b/llvm/test/CodeGen/AMDGPU/required-export-priority.ll
@@ -1,7 +1,7 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -mtriple=amdgpu11.00 -amdgpu-enable-vopd=0 < %s | FileCheck -check-prefixes=GCN,GFX11 %s
-; RUN: llc -mtriple=amdgpu11.50 -amdgpu-enable-vopd=0 < %s | FileCheck -check-prefixes=GCN,GFX1150 %s
-; RUN: llc -mtriple=amdgpu11.70 -amdgpu-enable-vopd=0 < %s | FileCheck -check-prefixes=GCN,GFX11 %s
+; RUN: llc -mtriple=amdgpu11.00 -amdgpu-enable-vopd=0 -disable-promote-alloca-to-vector < %s | FileCheck -check-prefixes=GCN,GFX11 %s
+; RUN: llc -mtriple=amdgpu11.50 -amdgpu-enable-vopd=0 -disable-promote-alloca-to-vector < %s | FileCheck -check-prefixes=GCN,GFX1150 %s
+; RUN: llc -mtriple=amdgpu11.70 -amdgpu-enable-vopd=0 -disable-promote-alloca-to-vector < %s | FileCheck -check-prefixes=GCN,GFX11 %s
define amdgpu_ps void @test_export_zeroes_f32() #0 {
; GFX11-LABEL: test_export_zeroes_f32:
diff --git a/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll b/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll
index 4d33d0020fff0..9b523a8d7b97f 100644
--- a/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll
+++ b/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -O1 -mtriple=amdgpu9.0a-amd-amdhsa < %s | FileCheck %s
+; RUN: llc -O1 -mtriple=amdgpu9.0a-amd-amdhsa -disable-promote-alloca-to-vector < %s | FileCheck %s
define void @sgpr_scavenge_fi_stack_id(double %input, i1 %enter_fma_path, i1 %repeat_outer_loop, i1 %enter_sgpr_loop, i1 %enter_inner_loop, i1 %exit_inner_loop, i1 %zero_fma_result) {
; CHECK-LABEL: sgpr_scavenge_fi_stack_id:
>From a3c74fa62b607d0fa551e3f6ca01870cd5959206 Mon Sep 17 00:00:00 2001
From: Ruiling Song <ruiling.song at amd.com>
Date: Thu, 10 Sep 2026 16:58:35 +0800
Subject: [PATCH 2/9] Address review comments:
1. depends on getUnderlyingObjects to collect merged alloca
2. Refine the Link propagation based on a Map and add a test for link
propagation.
3. Change `Uses` to unordered set to handle duplication easily.
4. Fix UTC tests issues.
5. Change `UsersToRemove` to SmallSetVector to handle duplicated PHIs.
---
.../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 302 +++++++-----------
.../CodeGen/AMDGPU/promote-alloca-scoring.ll | 27 +-
.../AMDGPU/promote-alloca-vector-phi.ll | 46 ++-
.../Inputs/amdgpu_asm.ll | 2 +-
.../Inputs/amdgpu_asm.ll.expected | 2 +-
.../Inputs/amdgpu_generated_funcs.ll | 2 +-
...dgpu_generated_funcs.ll.generated.expected | 2 +-
...pu_generated_funcs.ll.nogenerated.expected | 2 +-
.../Inputs/amdgpu_isel.ll | 2 +-
.../Inputs/amdgpu_isel.ll.expected | 2 +-
10 files changed, 187 insertions(+), 202 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index aea62360d9536..7299ca3ffc3ea 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -88,16 +88,15 @@ static cl::opt<unsigned>
// We support vector indices of the form ((A * stride) >> shift) + B
// VarIndex is A, VarMul is stride, VarShift is shift and ConstIndex is B. All
-// parts are optional.
+// parts are optional. When BasePtr does not point to Alloca object, the final
+// index will be (index_of_BasePtr + index_of_gep).
struct GEPToVectorIndex {
WeakTrackingVH VarIndex = nullptr; // defaults to 0
ConstantInt *VarMul = nullptr; // defaults to 1
ConstantInt *VarShift = nullptr; // defaults to 0
ConstantInt *ConstIndex = nullptr; // defaults to 0
Value *Full = nullptr;
- // The root pointer this GEP chain is based on: either a group member alloca
- // (constant lane) or a pointer phi/select of members (dynamic lane). The
- // final lane index is (this GEP's within-object offset) + lane(BasePtr).
+ // The root pointer this GEP chain is based on.
Value *BasePtr = nullptr;
};
@@ -108,19 +107,16 @@ struct MemTransferInfo {
// Analysis for planning the different strategies of alloca promotion.
//
-// An AllocaAnalysis represents a *group* of one or more allocas that must be
-// promoted together because their pointers are merged by a phi or select. The
-// common case is a singleton group (Members == {Alloca}). When promoting to
-// vector, the whole group becomes a single combined vector value; each member
-// occupies a contiguous lane range starting at Vector.BaseLane[member].
+// An AllocaAnalysis may represent one alloca or a group of allocas if they need
+// to be promoted together.
struct AllocaAnalysis {
AllocaInst *Alloca = nullptr; // Primary member (lane 0).
DenseSet<Value *> Pointers;
- SmallVector<Use *> Uses;
+ SmallDenseSet<Use *> Uses;
unsigned Score = 0;
// True if some phi/select merges this alloca's pointer with null or a
// non-alloca object. Such merges are still fine for LDS promotion, but they
- // disable vector grouping (we can't assign the other operand a lane).
+ // disable vector promotion.
bool HaveUnpromotableMerge = false;
// All member allocas of this group, in lane order (Members[0] == Alloca).
@@ -137,12 +133,7 @@ struct AllocaAnalysis {
// pointer phis terminate and are only materialized once.
DenseMap<Value *, Value *> IndexCache;
SmallVector<Instruction *> Worklist;
- SmallVector<Instruction *> UsersToRemove;
- // Pointer phis/selects that merge group members. Replaced by parallel lane
- // -index phis/selects and removed (RAUW poison) after promotion. A
- // SetVector because a merge is a use of multiple members and would
- // otherwise be recorded once per member.
- SmallSetVector<Instruction *, 4> PtrMerges;
+ SmallSetVector<Instruction *, 8> UsersToRemove;
MapVector<GetElementPtrInst *, GEPToVectorIndex> GEPVectorIdx;
MapVector<MemTransferInst *, MemTransferInfo> TransferInfo;
} Vector;
@@ -315,55 +306,14 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
SmallVector<Instruction *, 4> WorkList({AA.Alloca});
- // Classify the "other" operand of a pointer phi/select (i.e. an operand not
- // derived from Cur). Returns false if the merge makes the alloca entirely
- // unpromotable (merged with an unknown object). Otherwise records links to
- // any other allocas (for grouping).
- //
- // The operand may itself be a chain of phis/selects (e.g. a phi of a phi), so
- // we look through them to find the set of underlying root allocas. Visited
- // guards against cyclic pointer phis.
SmallPtrSet<Value *, 8> Visited;
- std::function<bool(Value *)> ClassifyMergeOperand = [&](Value *Other) -> bool {
- Other = Other->stripPointerCasts();
- if (AA.Pointers.contains(Other))
- return true; // Derived from the same pointer.
- if (!Visited.insert(Other).second)
- return true; // Already classified (or on a phi cycle).
-
- if (isa<ConstantPointerNull, ConstantAggregateZero>(Other)) {
- // Fine for LDS (null stays null), but we can't give it a vector lane.
- AA.HaveUnpromotableMerge = true;
- return true;
- }
-
- Value *Obj = getUnderlyingObject(Other);
- // getUnderlyingObject looks through GEPs/casts but not phi/select; recurse
- // through those to reach the root allocas.
- if (auto *Phi = dyn_cast<PHINode>(Obj)) {
- for (Value *In : Phi->incoming_values())
- if (!ClassifyMergeOperand(In))
- return false;
- return true;
- }
- if (auto *SI = dyn_cast<SelectInst>(Obj)) {
- return ClassifyMergeOperand(SI->getTrueValue()) &&
- ClassifyMergeOperand(SI->getFalseValue());
- }
- auto *OtherAI = dyn_cast<AllocaInst>(Obj);
- if (!OtherAI)
- return false;
-
- if (OtherAI != AA.Alloca)
- AA.Links.push_back(OtherAI);
- return true;
- };
-
+ // Other allocas which are merged by phi/select with current alloca.
+ SmallSetVector<AllocaInst *, 4> MergedAllocas;
while (!WorkList.empty()) {
auto *Cur = WorkList.pop_back_val();
- if (find(AA.Pointers, Cur) != AA.Pointers.end())
+ if (!Visited.insert(Cur).second)
continue;
- AA.Pointers.insert(Cur);
+
for (auto &U : Cur->uses()) {
auto *Inst = cast<Instruction>(U.getUser());
if (isa<StoreInst>(Inst)) {
@@ -371,31 +321,32 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
return RejectUser(Inst, "pointer escapes via store");
}
}
- AA.Uses.push_back(&U);
+ AA.Uses.insert(&U);
if (isa<GetElementPtrInst>(U.getUser())) {
WorkList.push_back(Inst);
- } else if (auto *SI = dyn_cast<SelectInst>(Inst)) {
- // A select may merge this alloca's pointer with another promotable
- // alloca (recorded as a link for grouping), with the same alloca, or
- // with null. Anything else is rejected.
- if (!ClassifyMergeOperand(SI->getTrueValue()) ||
- !ClassifyMergeOperand(SI->getFalseValue()))
- return RejectUser(Inst, "select from incompatible alloca");
- WorkList.push_back(Inst);
- } else if (auto *Phi = dyn_cast<PHINode>(Inst)) {
- // Repeat for phis. Every incoming value must be derived from a
- // promotable alloca (this one or a linked one).
- bool AllOk = true;
- for (Value *Incoming : Phi->incoming_values())
- AllOk &= ClassifyMergeOperand(Incoming);
- if (!AllOk)
- return RejectUser(Inst, "phi with incompatible alloca");
-
- WorkList.push_back(Inst);
+ } else if (isa<SelectInst, PHINode>(Inst)) {
+ SmallVector<const Value *> BaseObjs;
+ getUnderlyingObjects(Inst, BaseObjs, &LI);
+
+ for (auto *Obj : BaseObjs) {
+ if (auto *BaseAlloca = dyn_cast<AllocaInst>(Obj)) {
+ if (BaseAlloca != AA.Alloca) {
+ MergedAllocas.insert(const_cast<AllocaInst *>(BaseAlloca));
+ }
+ WorkList.push_back(Inst);
+ } else if (isa<ConstantPointerNull, ConstantAggregateZero>(Obj)) {
+ AA.HaveUnpromotableMerge = true;
+ WorkList.push_back(Inst);
+ } else {
+ return RejectUser(Inst, "phi/select with unkown object");
+ }
+ }
}
}
}
+ for (auto *Alloca : MergedAllocas)
+ AA.Links.push_back(Alloca);
return true;
}
@@ -447,72 +398,72 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
: (MaxVGPRs * 32)) /
VGPRBudgetRatio;
- std::vector<AllocaAnalysis> Allocas;
- {
- // Collect uses for every candidate alloca independently. This also
- // records phi/select links between distinct allocas (see collectAllocaUses).
- std::vector<AllocaAnalysis> Raw;
- DenseMap<AllocaInst *, int> Index; // -> position in Raw, or -1 if rejected.
- for (Instruction &I : F.getEntryBlock()) {
- auto *AI = dyn_cast<AllocaInst>(&I);
- // Array allocations are probably not worth handling, since an allocation
- // of the array type is the canonical form.
- if (!AI || !AI->isStaticAlloca() || AI->isArrayAllocation())
- continue;
- LLVM_DEBUG(dbgs() << "Analyzing: " << *AI << '\n');
- AllocaAnalysis AA{AI};
- if (!collectAllocaUses(AA)) {
- Index[AI] = -1;
- continue;
- }
- Index[AI] = Raw.size();
- Raw.push_back(std::move(AA));
- }
+ SmallMapVector<AllocaInst *, AllocaAnalysis, 4> AllocaAnalysisMap;
+ for (Instruction &I : F.getEntryBlock()) {
+ auto *AI = dyn_cast<AllocaInst>(&I);
+ // Array allocations are probably not worth handling, since an allocation
+ // of the array type is the canonical form.
+ if (!AI || !AI->isStaticAlloca() || AI->isArrayAllocation())
+ continue;
+ LLVM_DEBUG(dbgs() << "Analyzing: " << *AI << '\n');
+ AllocaAnalysis AA{AI};
+ if (!collectAllocaUses(AA))
+ continue;
+ AllocaAnalysisMap.insert({AI, std::move(AA)});
+ }
- for (int I = 0, E = Raw.size(); I != E; ++I) {
- bool Promotable = true;
- for (AllocaInst *Linked : Raw[I].Links) {
- // If this alloca is not promotable, the whole group is not promotable
- // as well.
- if (Index[Linked] == -1) {
- // The whole group can not be promoted
- Promotable = false;
- break;
- }
- }
+ std::vector<AllocaAnalysis> Allocas;
+ for (auto &AA : AllocaAnalysisMap.values()) {
+ if (AA.Uses.empty())
+ continue;
- if (Promotable && !Raw[I].Uses.empty()) {
- for (AllocaInst *Linked : Raw[I].Links) {
- // We have an linked alloca which appears earlier, this is not leader.
- //
- assert(Index[Linked] > I && "This should be leader alloca\n");
- // Pull from the other member alloca into the current alloca, which
- // is the leader.
- AllocaAnalysis &LeaderAA = Raw[I];
- AllocaAnalysis &MemberAA = Raw[Index[Linked]];
-
- // Merge the uses into the leader alloca.
- append_range(LeaderAA.Uses, MemberAA.Uses);
- // Clear member uses so that it will be skipped later.
- MemberAA.Uses.clear();
-
- LeaderAA.Pointers.insert_range(MemberAA.Pointers);
- LeaderAA.Members.push_back(MemberAA.Alloca);
- LeaderAA.HaveUnpromotableMerge |= MemberAA.HaveUnpromotableMerge;
- }
- // We only need to process leader alloca for later steps.
- Allocas.push_back(Raw[I]);
+ bool Promotable = true;
+ SmallVector<AllocaInst *> Worklist;
+ LLVM_DEBUG(dbgs() << "Process leader alloca " << *AA.Alloca << '\n');
+ Worklist.append(AA.Links);
+ // Pull the information from links into the leader alloca.
+ while (!Worklist.empty()) {
+ auto *CurLink = Worklist.pop_back_val();
+ auto *LinkIter = AllocaAnalysisMap.find(CurLink);
+ if (LinkIter != AllocaAnalysisMap.end()) {
+ AllocaAnalysis &LinkAA = LinkIter->second;
+ // Skip if the linked alloca was done or points to the leader alloca
+ // itself.
+ if (LinkAA.Uses.empty() || LinkAA.Alloca == AA.Alloca)
+ continue;
+
+ LLVM_DEBUG({
+ dbgs() << " Process link: " << *LinkAA.Alloca << '\n';
+ for (auto *X : LinkAA.Uses)
+ dbgs() << " Add User " << *X->getUser() << '\n';
+ });
+
+ AA.Uses.insert_range(LinkAA.Uses);
+ LinkAA.Uses.clear();
+ AA.HaveUnpromotableMerge |= LinkAA.HaveUnpromotableMerge;
+ AA.Members.push_back(LinkAA.Alloca);
+
+ // Add indirect links to worklist, so that their info are properly
+ // propagated to the leader alloca.
+ Worklist.append(LinkAA.Links);
+ LinkAA.Links.clear();
+ } else {
+ LLVM_DEBUG(dbgs() << " The alloca is not promotable\n");
+ Promotable = false;
+ break;
}
}
- }
- // Allocas that are merged by phi/select would have duplicated Use entries,
- // remove them while preserving order.
- for (AllocaAnalysis &AA : Allocas) {
- if (!AA.isGroup())
- continue;
- SmallPtrSet<Use *, 16> Seen;
- llvm::erase_if(AA.Uses, [&](Use *U) { return !Seen.insert(U).second; });
+ if (Promotable) {
+ sort(AA.Members, [](AllocaInst *A, AllocaInst *B) -> bool {
+ return A->comesBefore(B);
+ });
+ LLVM_DEBUG({
+ for (auto *M : AA.Members)
+ dbgs() << " Members: " << *M << '\n';
+ });
+ Allocas.push_back(AA);
+ }
}
for (AllocaAnalysis &AA : Allocas) {
@@ -606,15 +557,11 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
if (Cached != AA.Vector.IndexCache.end())
return Cached->second;
- const auto BaseLaneConst = [&](AllocaInst *Member) -> Value * {
- return ConstantInt::get(Type::getInt32Ty(Ctx),
- AA.Vector.BaseLane.lookup(Member));
- };
-
// A pointer that is directly one of the member allocas indexes lane
// BaseLane[member].
if (auto *AI = dyn_cast<AllocaInst>(Ptr)) {
- Value *Idx = BaseLaneConst(AI);
+ Value *Idx =
+ ConstantInt::get(Type::getInt32Ty(Ctx), AA.Vector.BaseLane.lookup(AI));
AA.Vector.IndexCache[Ptr] = Idx;
return Idx;
}
@@ -623,8 +570,8 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
// before recursing so self-referential phis terminate.
if (auto *Phi = dyn_cast<PHINode>(Ptr)) {
IRBuilder<> B(Phi);
- PHINode *IdxPhi =
- B.CreatePHI(B.getInt32Ty(), Phi->getNumIncomingValues(), "promotealloca.idx");
+ PHINode *IdxPhi = B.CreatePHI(B.getInt32Ty(), Phi->getNumIncomingValues(),
+ "promotealloca.idx");
AA.Vector.IndexCache[Ptr] = IdxPhi;
for (unsigned I = 0, E = Phi->getNumIncomingValues(); I != E; ++I)
IdxPhi->addIncoming(calculateVectorIndex(Phi->getIncomingValue(I), AA),
@@ -691,6 +638,11 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
static std::optional<GEPToVectorIndex>
computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaAnalysis &AA,
Type *VecElemTy, const DataLayout &DL) {
+ auto &GEPVectorIndexMap = AA.Vector.GEPVectorIdx;
+ auto *IndexIter = GEPVectorIndexMap.find(GEP);
+ if (IndexIter != GEPVectorIndexMap.end())
+ return IndexIter->second;
+
// TODO: Extracting a "multiple of X" from a GEP might be a useful generic
// helper.
LLVMContext &Ctx = GEP->getContext();
@@ -725,8 +677,7 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaAnalysis &AA,
}
// This should either points to the alloca or known phi/select.
- assert(isa<AllocaInst>(CurPtr) ||
- (isa<PHINode, SelectInst>(CurPtr) /*&& AA.Pointers.contains(CurPtr)*/));
+ assert(isa<AllocaInst>(CurPtr) || (isa<PHINode, SelectInst>(CurPtr)));
int64_t VecElemSize = DL.getTypeAllocSize(VecElemTy);
if (VarOffsets.size() > 1)
@@ -746,8 +697,10 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaAnalysis &AA,
Result.ConstIndex = ConstantInt::get(Ctx, IndexQuot.sextOrTrunc(BW));
// If there are no variable offsets, only a constant offset, then we're done.
- if (VarOffsets.empty())
+ if (VarOffsets.empty()) {
+ GEPVectorIndexMap[GEP] = Result;
return Result;
+ }
// Scale is the stride in the (A * stride) part. Check that there is only one
// variable offset and extract the scale factor.
@@ -792,6 +745,7 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaAnalysis &AA,
Result.VarShift = ConstantInt::get(Ctx, APInt(BW, Log2_64(Divisor)));
}
+ GEPVectorIndexMap[GEP] = Result;
return Result;
}
@@ -1214,8 +1168,8 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
// Pointer phis/selects merging group members are handled by rewriting them
// to lane-index phis/selects during promotion.
- if (isa<PHINode, SelectInst>(Inst) && Inst->getType()->isPointerTy()) {
- AA.Vector.PtrMerges.insert(Inst);
+ if (isa<PHINode, SelectInst>(Inst)) {
+ AA.Vector.UsersToRemove.insert(Inst);
continue;
}
@@ -1258,8 +1212,7 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
if (!Index)
return RejectUser(Inst, "cannot compute vector index for GEP");
- AA.Vector.GEPVectorIdx[GEP] = std::move(Index.value());
- AA.Vector.UsersToRemove.push_back(Inst);
+ AA.Vector.UsersToRemove.insert(Inst);
continue;
}
@@ -1284,12 +1237,13 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
if (Ptr == AA.Alloca)
return ConstantInt::get(Ptr->getContext(), APInt(32, 0));
- GetElementPtrInst *GEP = cast<GetElementPtrInst>(Ptr);
- const auto &GEPI = AA.Vector.GEPVectorIdx.find(GEP)->second;
- if (GEPI.VarIndex)
+ auto Index = computeGEPToVectorIndex(cast<GetElementPtrInst>(Ptr), AA,
+ VecEltTy, DL);
+
+ if (!Index || Index->VarIndex || Index->BasePtr != AA.Alloca)
return nullptr;
- if (GEPI.ConstIndex)
- return GEPI.ConstIndex;
+ if (Index->ConstIndex)
+ return Index->ConstIndex;
return ConstantInt::get(Ptr->getContext(), APInt(32, 0));
};
@@ -1324,14 +1278,14 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
if (isAssumeLikeIntrinsic(Inst)) {
if (!Inst->use_empty())
return RejectUser(Inst, "assume-like intrinsic cannot have any users");
- AA.Vector.UsersToRemove.push_back(Inst);
+ AA.Vector.UsersToRemove.insert(Inst);
continue;
}
if (isa<ICmpInst>(Inst) && all_of(Inst->users(), [](User *U) {
return isAssumeLikeIntrinsic(cast<Instruction>(U));
})) {
- AA.Vector.UsersToRemove.push_back(Inst);
+ AA.Vector.UsersToRemove.insert(Inst);
continue;
}
@@ -1420,20 +1374,10 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
I->eraseFromParent();
}
- // The pointer phis/selects merging group members are now dead. Drop them
- // before the GEPs they reference. RAUW with poison first so chained merges
- // can be erased in any order.
- for (Instruction *I : AA.Vector.PtrMerges)
- I->replaceAllUsesWith(PoisonValue::get(I->getType()));
- for (Instruction *I : AA.Vector.PtrMerges) {
- assert(I->use_empty());
- I->eraseFromParent();
- }
-
// Delete all the users that are known to be removeable.
- for (Instruction *I : reverse(AA.Vector.UsersToRemove)) {
- I->dropDroppableUses();
- assert(I->use_empty());
+ // Replace the uses with poison first so they can be deleted in any order.
+ for (Instruction *I : AA.Vector.UsersToRemove) {
+ I->replaceAllUsesWith(PoisonValue::get(I->getType()));
I->eraseFromParent();
}
@@ -1985,8 +1929,8 @@ bool AMDGPUPromoteAllocaImpl::tryPromoteAllocaToLDS(
case Intrinsic::invariant_end:
case Intrinsic::launder_invariant_group:
case Intrinsic::strip_invariant_group: {
- assert(Intr->getArgOperand(Intr->arg_size() - 1)->getType() == NewPtrTy &&
- "pointer operand should already have been promoted");
+ // Since the worklist can be in any order, the argument may still be the
+ // old type. Its type will be fixed when processing its definition.
Function *NewF = Intrinsic::getOrInsertDeclaration(
Intr->getModule(), Intr->getIntrinsicID(), NewPtrTy);
Intr->mutateType(NewF->getReturnType());
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll
index 473fa92547a43..922287a10b73c 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll
@@ -1,16 +1,14 @@
; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -debug-only=amdgpu-promote-alloca -amdgpu-promote-alloca-to-vector-limit=512 -passes=amdgpu-promote-alloca %s -o - 2>&1 | FileCheck %s
; REQUIRES: asserts
-; CHECK-LABEL: Analyzing: %simpleuser = alloca [4 x i64], align 4, addrspace(5)
-; CHECK-NEXT: Analyzing: %manyusers = alloca [4 x i64], align 4, addrspace(5)
-; CHECK-NEXT: Scoring: %simpleuser = alloca [4 x i64], align 4, addrspace(5)
+; CHECK-LABEL: Scoring: %simpleuser = alloca [4 x i64], align 4, addrspace(5)
; CHECK-NEXT: [+1]: store i64 42, ptr addrspace(5) %simpleuser, align 8
; CHECK-NEXT: => Final Score:1
; CHECK-NEXT: Scoring: %manyusers = alloca [4 x i64], align 4, addrspace(5)
-; CHECK-NEXT: [+1]: store i64 %v0.add, ptr addrspace(5) %manyusers.1, align 8
-; CHECK-NEXT: [+1]: %v0 = load i64, ptr addrspace(5) %manyusers.1, align 8
-; CHECK-NEXT: [+1]: store i64 %v1.add, ptr addrspace(5) %manyusers.2, align 8
-; CHECK-NEXT: [+1]: %v1 = load i64, ptr addrspace(5) %manyusers.2, align 8
+; CHECK-DAG: [+1]: store i64 %v0.add, ptr addrspace(5) %manyusers.1, align 8
+; CHECK-DAG: [+1]: %v0 = load i64, ptr addrspace(5) %manyusers.1, align 8
+; CHECK-DAG: [+1]: store i64 %v1.add, ptr addrspace(5) %manyusers.2, align 8
+; CHECK-DAG: [+1]: %v1 = load i64, ptr addrspace(5) %manyusers.2, align 8
; CHECK-NEXT: => Final Score:4
; CHECK-NEXT: Sorted Worklist:
; CHECK-NEXT: %manyusers = alloca [4 x i64], align 4, addrspace(5)
@@ -37,14 +35,13 @@ entry:
ret void
}
-; CHECK-LABEL: Analyzing: %stack = alloca [4 x i64], align 4, addrspace(5)
-; CHECK-NEXT: Scoring: %stack = alloca [4 x i64], align 4, addrspace(5)
-; CHECK-NEXT: [+5]: store i64 32, ptr addrspace(5) %stack, align 8
-; CHECK-NEXT: [+1]: store i64 42, ptr addrspace(5) %stack, align 8
-; CHECK-NEXT: [+9]: store i64 32, ptr addrspace(5) %stack.1, align 8
-; CHECK-NEXT: [+5]: %outer = load i64, ptr addrspace(5) %stack.1, align 8
-; CHECK-NEXT: [+1]: store i64 64, ptr addrspace(5) %stack.2, align 8
-; CHECK-NEXT: [+9]: %inner = load i64, ptr addrspace(5) %stack.2, align 8
+; CHECK-LABEL: Scoring: %stack = alloca [4 x i64], align 4, addrspace(5)
+; CHECK-DAG: [+5]: store i64 32, ptr addrspace(5) %stack, align 8
+; CHECK-DAG: [+1]: store i64 42, ptr addrspace(5) %stack, align 8
+; CHECK-DAG: [+9]: store i64 32, ptr addrspace(5) %stack.1, align 8
+; CHECK-DAG: [+5]: %outer = load i64, ptr addrspace(5) %stack.1, align 8
+; CHECK-DAG: [+1]: store i64 64, ptr addrspace(5) %stack.2, align 8
+; CHECK-DAG: [+9]: %inner = load i64, ptr addrspace(5) %stack.2, align 8
; CHECK-NEXT: => Final Score:30
define amdgpu_kernel void @loop_users_alloca(i1 %x, i2) #0 {
entry:
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
index b57874848e497..ab79cc2fd3119 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -S -mtriple=amdgpu-unknown-amdhsa -passes=amdgpu-promote-alloca-to-vector < %s | FileCheck %s
+; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca-to-vector < %s | FileCheck %s
; Tests for promotion of allocas whose pointers are merged by phis/selects,
@@ -38,6 +38,50 @@ join:
ret void
}
+define amdgpu_kernel void @phi_three_allocas(i1 %f, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_three_allocas(
+; CHECK-SAME: i1 [[F:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[A:%.*]] = freeze <3 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <3 x i32> [[A]], i32 1, i32 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <3 x i32> [[TMP0]], i32 2, i32 1
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <3 x i32> [[TMP1]], i32 3, i32 2
+; CHECK-NEXT: br i1 [[F]], label %[[BB_1:.*]], label %[[BB_2:.*]]
+; CHECK: [[BB_1]]:
+; CHECK-NEXT: br label %[[JOIN:.*]]
+; CHECK: [[BB_2]]:
+; CHECK-NEXT: br label %[[JOIN]]
+; CHECK: [[JOIN]]:
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX:%.*]] = phi i32 [ 0, %[[BB_1]] ], [ 1, %[[BB_2]] ]
+; CHECK-NEXT: [[PROMOTEALLOCA_IDX1:%.*]] = phi i32 [ 1, %[[BB_1]] ], [ 2, %[[BB_2]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = extractelement <3 x i32> [[TMP2]], i32 [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <3 x i32> [[TMP2]], i32 [[PROMOTEALLOCA_IDX1]]
+; CHECK-NEXT: [[L:%.*]] = add i32 [[TMP3]], [[TMP4]]
+; CHECK-NEXT: store i32 [[L]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %a = alloca i32, addrspace(5)
+ %b = alloca i32, addrspace(5)
+ %c = alloca i32, addrspace(5)
+ store i32 1, ptr addrspace(5) %a
+ store i32 2, ptr addrspace(5) %b
+ store i32 3, ptr addrspace(5) %c
+ br i1 %f, label %bb.1, label %bb.2
+bb.1:
+ br label %join
+bb.2:
+ br label %join
+join:
+ %p = phi ptr addrspace(5) [ %a, %bb.1 ], [ %b, %bb.2 ]
+ %q = phi ptr addrspace(5) [ %b, %bb.1 ], [ %c, %bb.2 ]
+ %l1 = load i32, ptr addrspace(5) %p
+ %l2 = load i32, ptr addrspace(5) %q
+ %l = add i32 %l1, %l2
+ store i32 %l, ptr addrspace(1) %out
+ ret void
+}
+
; GEP uses a (chained) pointer phi: phi-of-phi feeding a gep.
define amdgpu_kernel void @gep_uses_chained_phi(i1 %c, i1 %d, i32 %v, ptr addrspace(1) %out) {
; CHECK-LABEL: define amdgpu_kernel void @gep_uses_chained_phi(
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll
index 8b88768f6ae7a..74a7398be0558 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll
@@ -1,4 +1,4 @@
-; RUN: llc -mtriple=amdgcn-amd-amdhsa < %s | FileCheck %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -disable-promote-alloca-to-vector < %s | FileCheck %s
define i64 @i64_test(i64 %i) nounwind readnone {
%loc = alloca i64, addrspace(5)
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll.expected b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll.expected
index 53a164ed71759..6be3b24a735f3 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll.expected
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -mtriple=amdgcn-amd-amdhsa < %s | FileCheck %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -disable-promote-alloca-to-vector < %s | FileCheck %s
define i64 @i64_test(i64 %i) nounwind readnone {
; CHECK-LABEL: i64_test:
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll
index 6381aaf63c84e..42d2eb3984f10 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll
@@ -1,4 +1,4 @@
-; RUN: llc -enable-machine-outliner -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
+; RUN: llc -enable-machine-outliner -disable-promote-alloca-to-vector -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
; NOTE: Machine outliner doesn't run.
@x = dso_local global i32 0, align 4
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.generated.expected b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.generated.expected
index 0a85133152679..02d207f5c8609 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.generated.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.generated.expected
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --include-generated-funcs
-; RUN: llc -enable-machine-outliner -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
+; RUN: llc -enable-machine-outliner -disable-promote-alloca-to-vector -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
; NOTE: Machine outliner doesn't run.
@x = dso_local global i32 0, align 4
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.nogenerated.expected b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.nogenerated.expected
index df156b1b2e1b4..c913aeb16fbf8 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.nogenerated.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.nogenerated.expected
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -enable-machine-outliner -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
+; RUN: llc -enable-machine-outliner -disable-promote-alloca-to-vector -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
; NOTE: Machine outliner doesn't run.
@x = dso_local global i32 0, align 4
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll
index eabcbcbc79528..8f3e3e1f5eda9 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll
@@ -1,4 +1,4 @@
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -stop-after=finalize-isel -debug-only=isel -o /dev/null %s 2>&1 | FileCheck %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -disable-promote-alloca-to-vector -stop-after=finalize-isel -debug-only=isel -o /dev/null %s 2>&1 | FileCheck %s
define i64 @i64_test(i64 %i) nounwind readnone {
%loc = alloca i64, addrspace(5)
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll.expected b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll.expected
index ba42798c5c9be..ada56f9b08d12 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll.expected
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -stop-after=finalize-isel -debug-only=isel -o /dev/null %s 2>&1 | FileCheck %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -disable-promote-alloca-to-vector -stop-after=finalize-isel -debug-only=isel -o /dev/null %s 2>&1 | FileCheck %s
define i64 @i64_test(i64 %i) nounwind readnone {
; CHECK-LABEL: i64_test:
>From 0d6770bb9d2d09f3351652ddc73fc4f147fc3b24 Mon Sep 17 00:00:00 2001
From: Ruiling Song <ruiling.song at amd.com>
Date: Fri, 11 Sep 2026 15:23:08 +0800
Subject: [PATCH 3/9] Fix build failures
---
llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 2 +-
.../AMDGPU/promote-alloca-proper-value-replacement.ll | 7 +++----
2 files changed, 4 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index 48a793b0174a2..dcad3bdb5df4f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -1149,7 +1149,6 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
ElemCnt = cast<FixedVectorType>(MemberTy)->getNumElements();
}
-
AA.Vector.BaseLane[Member] = TotalElems;
TotalElems += ElemCnt;
}
@@ -1388,6 +1387,7 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
// Delete all the users that are known to be removeable.
// Replace the uses with poison first so they can be deleted in any order.
for (Instruction *I : AA.Vector.UsersToRemove) {
+ I->dropDroppableUses();
I->replaceAllUsesWith(PoisonValue::get(I->getType()));
I->eraseFromParent();
}
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-proper-value-replacement.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-proper-value-replacement.ll
index 8668c4721faa8..befd06ea429b3 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-proper-value-replacement.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-proper-value-replacement.ll
@@ -7,8 +7,8 @@ define void @alloca_value_cross_reference() {
; CHECK-NEXT: [[HIT_ORDERED:%.*]] = freeze <4 x float> poison
; CHECK-NEXT: [[HIT_INDEX:%.*]] = freeze <4 x i32> poison
; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x i32> [[HIT_INDEX]], i32 1, i32 0
-; CHECK-NEXT: br [[DOTLR_PH5:label %.*]]
-; CHECK: [[_LR_PH5:.*:]]
+; CHECK-NEXT: br label %[[DOTLR_PH5:.*]]
+; CHECK: [[DOTLR_PH5]]:
; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x i32> [[TMP0]], i32 0
; CHECK-NEXT: [[TMP2:%.*]] = insertelement <4 x float> [[HIT_ORDERED]], float 0.000000e+00, i32 [[TMP1]]
; CHECK-NEXT: ret void
@@ -41,10 +41,9 @@ define half @forwarded_load_across_blocks() {
; CHECK-NEXT: [[ARR:%.*]] = freeze <4 x half> poison
; CHECK-NEXT: br label %[[BB2:.*]]
; CHECK: [[BB2]]:
-; CHECK-NEXT: [[TMP0:%.*]] = freeze <4 x half> <half 1.000000e+00, half 2.000000e+00, half 3.000000e+00, half 4.000000e+00>
; CHECK-NEXT: br label %[[BB3:.*]]
; CHECK: [[BB3]]:
-; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x half> [[TMP0]], i32 0
+; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x half> <half 1.000000e+00, half 2.000000e+00, half 3.000000e+00, half 4.000000e+00>, i32 0
; CHECK-NEXT: ret half [[TMP1]]
;
entry:
>From da8d4311ea7e9ba285d7be1b3f39d57a0ab40af9 Mon Sep 17 00:00:00 2001
From: Ruiling Song <ruiling.song at amd.com>
Date: Mon, 14 Sep 2026 10:48:06 +0800
Subject: [PATCH 4/9] Replace AA.Alloca with AA.getLeaderAlloca()
---
.../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 93 ++++++++++---------
1 file changed, 51 insertions(+), 42 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index dcad3bdb5df4f..f8ea7b15d08b6 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -110,7 +110,6 @@ struct MemTransferInfo {
// An AllocaAnalysis may represent one alloca or a group of allocas if they need
// to be promoted together.
struct AllocaAnalysis {
- AllocaInst *Alloca = nullptr; // Primary member (lane 0).
DenseSet<Value *> Pointers;
SmallDenseSet<Use *> Uses;
unsigned Score = 0;
@@ -142,11 +141,12 @@ struct AllocaAnalysis {
SmallVector<User *> Worklist;
} LDS;
- explicit AllocaAnalysis(AllocaInst *Alloca) : Alloca(Alloca) {
- Members.push_back(Alloca);
- }
+ explicit AllocaAnalysis(AllocaInst *Alloca) { Members.push_back(Alloca); }
bool isGroup() const { return Members.size() > 1; }
+
+ // Returns the leader alloca if this is an alloca group.
+ AllocaInst *getLeaderAlloca() const { return Members[0]; }
};
// Shared implementation which can do both promotion to vector and to LDS.
@@ -304,7 +304,7 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
return false;
};
- SmallVector<Instruction *, 4> WorkList({AA.Alloca});
+ SmallVector<Instruction *, 4> WorkList({AA.getLeaderAlloca()});
SmallPtrSet<Value *, 8> Visited;
// Other allocas which are merged by phi/select with current alloca.
@@ -331,7 +331,7 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
for (auto *Obj : BaseObjs) {
if (auto *BaseAlloca = dyn_cast<AllocaInst>(Obj)) {
- if (BaseAlloca != AA.Alloca) {
+ if (BaseAlloca != AA.getLeaderAlloca()) {
MergedAllocas.insert(const_cast<AllocaInst *>(BaseAlloca));
}
WorkList.push_back(Inst);
@@ -351,7 +351,7 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
}
void AMDGPUPromoteAllocaImpl::scoreAlloca(AllocaAnalysis &AA) const {
- LLVM_DEBUG(dbgs() << "Scoring: " << *AA.Alloca << "\n");
+ LLVM_DEBUG(dbgs() << "Scoring: " << *AA.getLeaderAlloca() << "\n");
unsigned Score = 0;
// Increment score by one for each user + a bonus for users within loops.
for (auto *U : AA.Uses) {
@@ -419,21 +419,22 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
bool Promotable = true;
SmallVector<AllocaInst *> Worklist;
- LLVM_DEBUG(dbgs() << "Process leader alloca " << *AA.Alloca << '\n');
+ LLVM_DEBUG(dbgs() << "Process leader alloca " << *AA.getLeaderAlloca()
+ << '\n');
Worklist.append(AA.Links);
// Pull the information from links into the leader alloca.
while (!Worklist.empty()) {
auto *CurLink = Worklist.pop_back_val();
- auto *LinkIter = AllocaAnalysisMap.find(CurLink);
+ auto LinkIter = AllocaAnalysisMap.find(CurLink);
if (LinkIter != AllocaAnalysisMap.end()) {
AllocaAnalysis &LinkAA = LinkIter->second;
// Skip if the linked alloca was done or points to the leader alloca
// itself.
- if (LinkAA.Uses.empty() || LinkAA.Alloca == AA.Alloca)
+ if (LinkAA.Uses.empty() || &LinkAA == &AA)
continue;
LLVM_DEBUG({
- dbgs() << " Process link: " << *LinkAA.Alloca << '\n';
+ dbgs() << " Process link: " << *LinkAA.getLeaderAlloca() << '\n';
for (auto *X : LinkAA.Uses)
dbgs() << " Add User " << *X->getUser() << '\n';
});
@@ -441,7 +442,7 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
AA.Uses.insert_range(LinkAA.Uses);
LinkAA.Uses.clear();
AA.HaveUnpromotableMerge |= LinkAA.HaveUnpromotableMerge;
- AA.Members.push_back(LinkAA.Alloca);
+ AA.Members.push_back(LinkAA.getLeaderAlloca());
// Add indirect links to worklist, so that their info are properly
// propagated to the leader alloca.
@@ -487,7 +488,7 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
LLVM_DEBUG(
dbgs() << "Sorted Worklist:\n";
for (const auto &AA : Allocas)
- dbgs() << " " << *AA.Alloca << "\n";
+ dbgs() << " " << *AA.getLeaderAlloca() << "\n";
);
// clang-format on
@@ -514,7 +515,7 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
} else {
LLVM_DEBUG(dbgs() << "Alloca too big for vectorization (size:"
<< AllocaCost << ", budget:" << VectorizationBudget
- << "): " << *AA.Alloca << "\n");
+ << "): " << *AA.getLeaderAlloca() << "\n");
}
}
@@ -622,7 +623,7 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
// Offset by the lane of the root pointer this GEP is based on. For a member
// alloca this is its constant base lane; for a pointer phi/select it is the
// parallel lane-index value.
- if (I->second.BasePtr != AA.Alloca || AA.isGroup()) {
+ if (I->second.BasePtr != AA.getLeaderAlloca() || AA.isGroup()) {
Value *BaseIdx = calculateVectorIndex(I->second.BasePtr, AA);
if (auto *C = dyn_cast<ConstantInt>(BaseIdx); !C || !C->isZero())
Result = B.CreateAdd(Result, BaseIdx);
@@ -1201,8 +1202,8 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
Ptr = Ptr->stripPointerCasts();
// Alloca already accessed as vector for single alloca group case.
- if (!AA.isGroup() && Ptr == AA.Alloca &&
- DL.getTypeStoreSize(AA.Alloca->getAllocatedType()) ==
+ if (!AA.isGroup() && Ptr == AA.getLeaderAlloca() &&
+ DL.getTypeStoreSize(AA.getLeaderAlloca()->getAllocatedType()) ==
DL.getTypeStoreSize(AccessTy)) {
AA.Vector.Worklist.push_back(Inst);
continue;
@@ -1227,7 +1228,8 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
}
if (MemSetInst *MSI = dyn_cast<MemSetInst>(Inst);
- MSI && !AA.isGroup() && isSupportedMemset(MSI, AA.Alloca, DL)) {
+ MSI && !AA.isGroup() &&
+ isSupportedMemset(MSI, AA.getLeaderAlloca(), DL)) {
AA.Vector.Worklist.push_back(Inst);
continue;
}
@@ -1244,13 +1246,13 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
"not a multiple of the vector element size");
auto getConstIndexIntoAlloca = [&](Value *Ptr) -> ConstantInt * {
- if (Ptr == AA.Alloca)
+ if (Ptr == AA.getLeaderAlloca())
return ConstantInt::get(Ptr->getContext(), APInt(32, 0));
auto Index = computeGEPToVectorIndex(cast<GetElementPtrInst>(Ptr), AA,
VecEltTy, DL);
- if (!Index || Index->VarIndex || Index->BasePtr != AA.Alloca)
+ if (!Index || Index->VarIndex || Index->BasePtr != AA.getLeaderAlloca())
return nullptr;
if (Index->ConstIndex)
return Index->ConstIndex;
@@ -1313,9 +1315,11 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
}
void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
- LLVM_DEBUG(dbgs() << "Promoting to vectors: " << *AA.Alloca << '\n');
- LLVM_DEBUG(dbgs() << " type conversion: " << *AA.Alloca->getAllocatedType()
- << " -> " << *AA.Vector.Ty << '\n');
+ LLVM_DEBUG(dbgs() << "Promoting to vectors: " << *AA.getLeaderAlloca()
+ << '\n');
+ LLVM_DEBUG(dbgs() << " type conversion: "
+ << *AA.getLeaderAlloca()->getAllocatedType() << " -> "
+ << *AA.Vector.Ty << '\n');
const unsigned VecStoreSize = DL.getTypeStoreSize(AA.Vector.Ty);
Type *VecEltTy = AA.Vector.Ty->getElementType();
@@ -1326,14 +1330,14 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
SSAUpdater Updater;
Updater.Initialize(AA.Vector.Ty, "promotealloca");
- BasicBlock *EntryBB = AA.Alloca->getParent();
+ BasicBlock *EntryBB = AA.getLeaderAlloca()->getParent();
BasicBlock::iterator InitInsertPos =
- skipToNonAllocaInsertPt(*EntryBB, AA.Alloca->getIterator());
+ skipToNonAllocaInsertPt(*EntryBB, AA.getLeaderAlloca()->getIterator());
IRBuilder<> Builder(&*InitInsertPos);
Value *AllocaInitValue = Builder.CreateFreeze(PoisonValue::get(AA.Vector.Ty));
- AllocaInitValue->takeName(AA.Alloca);
+ AllocaInitValue->takeName(AA.getLeaderAlloca());
- Updater.AddAvailableValue(AA.Alloca->getParent(), AllocaInitValue);
+ Updater.AddAvailableValue(AA.getLeaderAlloca()->getParent(), AllocaInitValue);
// First handle the initial worklist, in basic block order.
//
@@ -1579,7 +1583,7 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToLDS(AllocaAnalysis &AA) const {
// Don't promote the alloca to LDS for shader calling conventions as the work
// item ID intrinsics are not supported for these calling conventions.
// Furthermore not all LDS is available for some of the stages.
- const Function &ContainingFunction = *AA.Alloca->getFunction();
+ const Function &ContainingFunction = *AA.getLeaderAlloca()->getFunction();
CallingConv::ID CC = ContainingFunction.getCallingConv();
switch (CC) {
@@ -1636,7 +1640,8 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToLDS(AllocaAnalysis &AA) const {
// Only promote a select if we know that the other select operand
// is from another pointer that will also be promoted.
if (ICmpInst *ICmp = dyn_cast<ICmpInst>(UseInst)) {
- if (!binaryOpIsDerivedFromSameAlloca(AA.Alloca, Use->get(), ICmp, 0, 1))
+ if (!binaryOpIsDerivedFromSameAlloca(AA.getLeaderAlloca(), Use->get(),
+ ICmp, 0, 1))
return;
// May need to rewrite constant operands.
@@ -1798,19 +1803,21 @@ bool AMDGPUPromoteAllocaImpl::hasSufficientLocalMem(const Function &F) {
bool AMDGPUPromoteAllocaImpl::tryPromoteAllocaToLDS(
AllocaAnalysis &AA, bool SufficientLDS,
SetVector<IntrinsicInst *> &DeferredIntrs) {
- LLVM_DEBUG(dbgs() << "Trying to promote to LDS: " << *AA.Alloca << '\n');
+ LLVM_DEBUG(dbgs() << "Trying to promote to LDS: " << *AA.getLeaderAlloca()
+ << '\n');
// Not likely to have sufficient local memory for promotion.
if (!SufficientLDS)
return false;
- IRBuilder<> Builder(AA.Alloca);
+ IRBuilder<> Builder(AA.getLeaderAlloca());
- const Function &ContainingFunction = *AA.Alloca->getParent()->getParent();
+ const Function &ContainingFunction =
+ *AA.getLeaderAlloca()->getParent()->getParent();
const AMDGPUSubtarget &ST = AMDGPUSubtarget::get(TM, ContainingFunction);
unsigned WorkGroupSize = ST.getFlatWorkGroupSizes(ContainingFunction).second;
- Align Alignment = AA.Alloca->getAlign();
+ Align Alignment = AA.getLeaderAlloca()->getAlign();
// FIXME: This computed padding is likely wrong since it depends on inverse
// usage order.
@@ -1819,7 +1826,8 @@ bool AMDGPUPromoteAllocaImpl::tryPromoteAllocaToLDS(
// could end up using more than the maximum due to alignment padding.
uint32_t NewSize = alignTo(CurrentLocalMemUsage, Alignment);
- std::optional<TypeSize> ElemSize = AA.Alloca->getAllocationSize(DL);
+ std::optional<TypeSize> ElemSize =
+ AA.getLeaderAlloca()->getAllocationSize(DL);
if (!ElemSize || ElemSize->isScalable())
return false;
TypeSize AllocSize = WorkGroupSize * *ElemSize;
@@ -1835,15 +1843,16 @@ bool AMDGPUPromoteAllocaImpl::tryPromoteAllocaToLDS(
LLVM_DEBUG(dbgs() << "Promoting alloca to local memory\n");
- Function *F = AA.Alloca->getFunction();
+ Function *F = AA.getLeaderAlloca()->getFunction();
- Type *GVTy = ArrayType::get(AA.Alloca->getAllocatedType(), WorkGroupSize);
+ Type *GVTy =
+ ArrayType::get(AA.getLeaderAlloca()->getAllocatedType(), WorkGroupSize);
GlobalVariable *GV = new GlobalVariable(
Mod, GVTy, false, GlobalValue::InternalLinkage, PoisonValue::get(GVTy),
- Twine(F->getName()) + Twine('.') + AA.Alloca->getName(), nullptr,
- GlobalVariable::NotThreadLocal, AMDGPUAS::LOCAL_ADDRESS);
+ Twine(F->getName()) + Twine('.') + AA.getLeaderAlloca()->getName(),
+ nullptr, GlobalVariable::NotThreadLocal, AMDGPUAS::LOCAL_ADDRESS);
GV->setUnnamedAddr(GlobalValue::UnnamedAddr::Global);
- GV->setAlignment(AA.Alloca->getAlign());
+ GV->setAlignment(AA.getLeaderAlloca()->getAlign());
Value *TCntY, *TCntZ;
@@ -1862,9 +1871,9 @@ bool AMDGPUPromoteAllocaImpl::tryPromoteAllocaToLDS(
Value *Indices[] = {Constant::getNullValue(Type::getInt32Ty(Context)), TID};
Value *Offset = Builder.CreateInBoundsGEP(GVTy, GV, Indices);
- AA.Alloca->mutateType(Offset->getType());
- AA.Alloca->replaceAllUsesWith(Offset);
- AA.Alloca->eraseFromParent();
+ AA.getLeaderAlloca()->mutateType(Offset->getType());
+ AA.getLeaderAlloca()->replaceAllUsesWith(Offset);
+ AA.getLeaderAlloca()->eraseFromParent();
PointerType *NewPtrTy = PointerType::get(Context, AMDGPUAS::LOCAL_ADDRESS);
>From 27a83a0b9300f2efb4364cf4d272406d2a70974b Mon Sep 17 00:00:00 2001
From: Ruiling Song <ruiling.song at amd.com>
Date: Mon, 14 Sep 2026 14:50:16 +0800
Subject: [PATCH 5/9] No auto usage
---
llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 7 ++++---
1 file changed, 4 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index f8ea7b15d08b6..f80998965fe8e 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -424,7 +424,7 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
Worklist.append(AA.Links);
// Pull the information from links into the leader alloca.
while (!Worklist.empty()) {
- auto *CurLink = Worklist.pop_back_val();
+ AllocaInst *CurLink = Worklist.pop_back_val();
auto LinkIter = AllocaAnalysisMap.find(CurLink);
if (LinkIter != AllocaAnalysisMap.end()) {
AllocaAnalysis &LinkAA = LinkIter->second;
@@ -460,7 +460,7 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
return A->comesBefore(B);
});
LLVM_DEBUG({
- for (auto *M : AA.Members)
+ for (AllocaInst *M : AA.Members)
dbgs() << " Members: " << *M << '\n';
});
Allocas.push_back(AA);
@@ -574,9 +574,10 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
PHINode *IdxPhi = B.CreatePHI(B.getInt32Ty(), Phi->getNumIncomingValues(),
"promotealloca.idx");
AA.Vector.IndexCache[Ptr] = IdxPhi;
- for (unsigned I = 0, E = Phi->getNumIncomingValues(); I != E; ++I)
+ for (unsigned I = 0, E = Phi->getNumIncomingValues(); I != E; ++I) {
IdxPhi->addIncoming(calculateVectorIndex(Phi->getIncomingValue(I), AA),
Phi->getIncomingBlock(I));
+ }
return IdxPhi;
}
>From d25845a95f077fe5dfb0916a5670a93387c2f547 Mon Sep 17 00:00:00 2001
From: Ruiling Song <ruiling.song at amd.com>
Date: Mon, 14 Sep 2026 14:50:58 +0800
Subject: [PATCH 6/9] Make the load/store alloca to volatile to stop promotion
Co-Authored-By: Claude
---
.../CodeGen/AMDGPU/frame-index-elimination.ll | 19 +--
.../memory-legalizer-store-infinite-loop.ll | 4 +-
.../AMDGPU/required-export-priority.ll | 21 +--
.../AMDGPU/sgpr-scavenge-fi-stack-id.ll | 11 +-
.../Inputs/amdgpu_asm.ll | 10 +-
.../Inputs/amdgpu_asm.ll.expected | 24 +--
.../Inputs/amdgpu_generated_funcs.ll | 48 +++---
...dgpu_generated_funcs.ll.generated.expected | 148 +++++++++++-------
...pu_generated_funcs.ll.nogenerated.expected | 148 +++++++++++-------
.../Inputs/amdgpu_isel.ll | 10 +-
.../Inputs/amdgpu_isel.ll.expected | 31 ++--
11 files changed, 272 insertions(+), 202 deletions(-)
diff --git a/llvm/test/CodeGen/AMDGPU/frame-index-elimination.ll b/llvm/test/CodeGen/AMDGPU/frame-index-elimination.ll
index 6f6a5c41cf11c..525d063a3c070 100644
--- a/llvm/test/CodeGen/AMDGPU/frame-index-elimination.ll
+++ b/llvm/test/CodeGen/AMDGPU/frame-index-elimination.ll
@@ -1,8 +1,8 @@
; RUN: llc -mtriple=amdgpu7.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds < %s | FileCheck -enable-var-scope -check-prefixes=GCN,CI,MUBUF %s
; RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds < %s | FileCheck -enable-var-scope -check-prefixes=GCN,GFX9,GFX9-MUBUF,MUBUF %s
; RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds -mattr=+enable-flat-scratch < %s | FileCheck -enable-var-scope -check-prefixes=GCN,GFX9,GFX9-FLATSCR %s
-; RUN: llc -mtriple=amdgpu11.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds -mattr=+real-true16 < %s | FileCheck --check-prefixes=GFX11-TRUE16 %s
-; RUN: llc -mtriple=amdgpu11.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds -mattr=-real-true16 < %s | FileCheck --check-prefixes=GFX11-FAKE16 %s
+; RUN: llc -mtriple=amdgpu11.00-amd-amdhsa -disable-promote-alloca-to-lds -mattr=+real-true16 < %s | FileCheck --check-prefixes=GFX11-TRUE16 %s
+; RUN: llc -mtriple=amdgpu11.00-amd-amdhsa -disable-promote-alloca-to-lds -mattr=-real-true16 < %s | FileCheck --check-prefixes=GFX11-FAKE16 %s
; Test that non-entry function frame indices are expanded properly to
; give an index relative to the scratch wave offset register
@@ -306,17 +306,18 @@ ret:
; GFX11-TRUE16-LABEL: tied_operand_test:
; GFX11-TRUE16: ; %bb.0: ; %entry
-; GFX11-TRUE16: scratch_load_d16_b16 [[LDRESULT:v[0-9]+]], off, off
-; GFX11-TRUE16: v_mov_b16_e32 [[C:v[0-9]]].{{(l|h)}}, 0x7b
-; GFX11-TRUE16-DAG: ds_store_b16 v{{[0-9]+}}, [[LDRESULT]] offset:10
+; GFX11-TRUE16: v_mov_b16_e32 [[C:v[0-9]+]].{{(l|h)}}, 0x7b
+; GFX11-TRUE16: scratch_load_d16_b16 [[LDRESULT:v[0-9]+]], off, off glc dlc
+; GFX11-TRUE16-DAG: ds_store_b16_d16_hi v{{[0-9]+}}, [[C]] offset:8
+; GFX11-TRUE16-DAG: ds_store_b16 v{{[0-9]+}}, [[LDRESULT]] offset:10
; GFX11-TRUE16-NEXT: s_endpgm
;
; GFX11-FAKE16-LABEL: tied_operand_test:
; GFX11-FAKE16: ; %bb.0: ; %entry
-; GFX11-FAKE16: scratch_load_u16 [[LDRESULT:v[0-9]+]], off, off
+; GFX11-FAKE16: scratch_load_u16 [[LDRESULT:v[0-9]+]], off, off glc dlc
; GFX11-FAKE16: v_dual_mov_b32 [[C:v[0-9]+]], 0x7b :: v_dual_mov_b32 v{{[0-9]+}}, s{{[0-9]+}}
-; GFX11-FAKE16-DAG: ds_store_b16 v{{[0-9]+}}, [[LDRESULT]] offset:10
; GFX11-FAKE16-DAG: ds_store_b16 v{{[0-9]+}}, [[C]] offset:8
+; GFX11-FAKE16-DAG: ds_store_b16 v{{[0-9]+}}, [[LDRESULT]] offset:10
; GFX11-FAKE16-NEXT: s_endpgm
define protected amdgpu_kernel void @tied_operand_test(i1 %c1, i1 %c2, i32 %val) {
entry:
@@ -324,8 +325,8 @@ entry:
%scratch1 = alloca i16, align 4, addrspace(5)
%first = select i1 %c1, ptr addrspace(5) %scratch0, ptr addrspace(5) %scratch1
%spec.select = select i1 %c2, ptr addrspace(5) %first, ptr addrspace(5) %scratch0
- %dead.load = load i16, ptr addrspace(5) %spec.select, align 2
- %scratch0.load = load i16, ptr addrspace(5) %scratch0, align 4
+ %dead.load = load volatile i16, ptr addrspace(5) %spec.select, align 2
+ %scratch0.load = load volatile i16, ptr addrspace(5) %scratch0, align 4
%add4 = add nuw nsw i32 %val, 4
%addr0 = getelementptr inbounds %struct0, ptr addrspace(3) @_ZZN0, i32 0, i32 0, i32 %add4, i32 0
store i16 123, ptr addrspace(3) %addr0, align 2
diff --git a/llvm/test/CodeGen/AMDGPU/memory-legalizer-store-infinite-loop.ll b/llvm/test/CodeGen/AMDGPU/memory-legalizer-store-infinite-loop.ll
index 880f8d3972ca1..55541952fc0bc 100644
--- a/llvm/test/CodeGen/AMDGPU/memory-legalizer-store-infinite-loop.ll
+++ b/llvm/test/CodeGen/AMDGPU/memory-legalizer-store-infinite-loop.ll
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -mtriple=amdgpu7.00--amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds < %s | FileCheck -check-prefix=GCN %s
+; RUN: llc -mtriple=amdgpu7.00--amdhsa < %s | FileCheck -check-prefix=GCN %s
; Effectively, check that the compile finishes; in the case
; of an infinite loop, llc toggles between merging 2 ST4s
@@ -38,7 +38,7 @@ bb3: ; No predecessors!
bb4: ; preds = %bb3, %bb
%tmp5 = phi ptr addrspace(5) [ %tmp1, %bb3 ], [ %tmp, %bb ]
- store double %tmp2, ptr addrspace(5) %tmp5, align 8
+ store volatile double %tmp2, ptr addrspace(5) %tmp5, align 8
br label %bb6
bb6: ; preds = %bb4, %bb
diff --git a/llvm/test/CodeGen/AMDGPU/required-export-priority.ll b/llvm/test/CodeGen/AMDGPU/required-export-priority.ll
index 4df215a3f6224..6e64a4575d082 100644
--- a/llvm/test/CodeGen/AMDGPU/required-export-priority.ll
+++ b/llvm/test/CodeGen/AMDGPU/required-export-priority.ll
@@ -1,7 +1,7 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -mtriple=amdgpu11.00 -amdgpu-enable-vopd=0 -disable-promote-alloca-to-vector < %s | FileCheck -check-prefixes=GCN,GFX11 %s
-; RUN: llc -mtriple=amdgpu11.50 -amdgpu-enable-vopd=0 -disable-promote-alloca-to-vector < %s | FileCheck -check-prefixes=GCN,GFX1150 %s
-; RUN: llc -mtriple=amdgpu11.70 -amdgpu-enable-vopd=0 -disable-promote-alloca-to-vector < %s | FileCheck -check-prefixes=GCN,GFX11 %s
+; RUN: llc -mtriple=amdgpu11.00 -amdgpu-enable-vopd=0 < %s | FileCheck -check-prefixes=GCN,GFX11 %s
+; RUN: llc -mtriple=amdgpu11.50 -amdgpu-enable-vopd=0 < %s | FileCheck -check-prefixes=GCN,GFX1150 %s
+; RUN: llc -mtriple=amdgpu11.70 -amdgpu-enable-vopd=0 < %s | FileCheck -check-prefixes=GCN,GFX11 %s
define amdgpu_ps void @test_export_zeroes_f32() #0 {
; GFX11-LABEL: test_export_zeroes_f32:
@@ -366,10 +366,10 @@ define amdgpu_ps void @test_export_across_store_load(i32 %idx, float %v) #0 {
; GFX11-NEXT: v_cndmask_b32_e32 v0, 16, v2, vcc_lo
; GFX11-NEXT: v_mov_b32_e32 v2, 0
; GFX11-NEXT: scratch_store_b32 v0, v1, off
-; GFX11-NEXT: scratch_load_b32 v0, off, off
+; GFX11-NEXT: scratch_load_b32 v0, off, off glc dlc
+; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_mov_b32_e32 v1, 1.0
; GFX11-NEXT: exp pos0, v2, v2, v2, v1 done
-; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: exp invalid_target_32, v0, v2, v1, v2
; GFX11-NEXT: exp invalid_target_33, v0, v2, v1, v2
; GFX11-NEXT: s_endpgm
@@ -383,15 +383,10 @@ define amdgpu_ps void @test_export_across_store_load(i32 %idx, float %v) #0 {
; GFX1150-NEXT: v_cndmask_b32_e32 v0, 16, v2, vcc_lo
; GFX1150-NEXT: v_mov_b32_e32 v2, 0
; GFX1150-NEXT: scratch_store_b32 v0, v1, off
-; GFX1150-NEXT: scratch_load_b32 v0, off, off
+; GFX1150-NEXT: scratch_load_b32 v0, off, off glc dlc
+; GFX1150-NEXT: s_waitcnt vmcnt(0)
; GFX1150-NEXT: v_mov_b32_e32 v1, 1.0
; GFX1150-NEXT: exp pos0, v2, v2, v2, v1 done
-; GFX1150-NEXT: s_setprio 0
-; GFX1150-NEXT: s_waitcnt_expcnt null, 0x0
-; GFX1150-NEXT: s_nop 0
-; GFX1150-NEXT: s_nop 0
-; GFX1150-NEXT: s_setprio 2
-; GFX1150-NEXT: s_waitcnt vmcnt(0)
; GFX1150-NEXT: exp invalid_target_32, v0, v2, v1, v2
; GFX1150-NEXT: exp invalid_target_33, v0, v2, v1, v2
; GFX1150-NEXT: s_setprio 0
@@ -404,7 +399,7 @@ define amdgpu_ps void @test_export_across_store_load(i32 %idx, float %v) #0 {
%data = select i1 %cmp, ptr addrspace(5) %data0, ptr addrspace(5) %data1
store float %v, ptr addrspace(5) %data, align 8
call void @llvm.amdgcn.exp.f32(i32 12, i32 15, float 0.0, float 0.0, float 0.0, float 1.0, i1 true, i1 false)
- %load0 = load float, ptr addrspace(5) %data0, align 8
+ %load0 = load volatile float, ptr addrspace(5) %data0, align 8
call void @llvm.amdgcn.exp.f32(i32 32, i32 15, float %load0, float 0.0, float 1.0, float 0.0, i1 false, i1 false)
call void @llvm.amdgcn.exp.f32(i32 33, i32 15, float %load0, float 0.0, float 1.0, float 0.0, i1 false, i1 false)
ret void
diff --git a/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll b/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll
index 9b523a8d7b97f..3adee6e4b668d 100644
--- a/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll
+++ b/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -O1 -mtriple=amdgpu9.0a-amd-amdhsa -disable-promote-alloca-to-vector < %s | FileCheck %s
+; RUN: llc -O1 -mtriple=amdgpu9.0a-amd-amdhsa < %s | FileCheck %s
define void @sgpr_scavenge_fi_stack_id(double %input, i1 %enter_fma_path, i1 %repeat_outer_loop, i1 %enter_sgpr_loop, i1 %enter_inner_loop, i1 %exit_inner_loop, i1 %zero_fma_result) {
; CHECK-LABEL: sgpr_scavenge_fi_stack_id:
@@ -179,11 +179,12 @@ define void @sgpr_scavenge_fi_stack_id(double %input, i1 %enter_fma_path, i1 %re
; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: s_not_b64 exec, exec
; CHECK-NEXT: v_mov_b32_e32 v7, vcc_hi
-; CHECK-NEXT: buffer_load_dword v8, v7, s[0:3], 0 offen
-; CHECK-NEXT: buffer_load_dword v9, v7, s[0:3], 0 offen offset:4
+; CHECK-NEXT: buffer_load_dword v8, v7, s[0:3], 0 offen glc
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: buffer_load_dword v9, v7, s[0:3], 0 offen offset:4 glc
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: v_mov_b32_e32 v7, vcc_lo
; CHECK-NEXT: s_mov_b32 vcc_lo, 1
-; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: v_mul_f64 v[8:9], v[4:5], v[8:9]
; CHECK-NEXT: buffer_store_dword v9, off, s[0:3], 0 offset:4
; CHECK-NEXT: buffer_store_dword v8, off, s[0:3], 0
@@ -270,7 +271,7 @@ inner_stack_loop: ; preds = %inner_stack_loop, %
%scaled_mul = fmul double %inf_mul, f0x40417E50A9DC6553
%nan_mul = fmul double %scaled_mul, +qnan
%private_gep = getelementptr [8 x i8], ptr addrspace(5) %private_slot, i32 %stack_offset
- %private_load = load double, ptr addrspace(5) %private_gep, align 1
+ %private_load = load volatile double, ptr addrspace(5) %private_gep, align 1
%spill_value = fmul double %nan_mul, %private_load
store double %spill_value, ptr addrspace(5) null, align 1
%indexed_null = getelementptr i8, ptr addrspace(5) null, i32 %stack_offset
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll
index 74a7398be0558..f80a7d6d4190b 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll
@@ -1,15 +1,15 @@
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -disable-promote-alloca-to-vector < %s | FileCheck %s
+; RUN: llc -mtriple=amdgpu7.00-amd-amdhsa < %s | FileCheck %s
define i64 @i64_test(i64 %i) nounwind readnone {
%loc = alloca i64, addrspace(5)
- %j = load i64, ptr addrspace(5) %loc
+ %j = load volatile i64, ptr addrspace(5) %loc
%r = add i64 %i, %j
ret i64 %r
}
define i64 @i32_test(i32 %i) nounwind readnone {
%loc = alloca i32, addrspace(5)
- %j = load i32, ptr addrspace(5) %loc
+ %j = load volatile i32, ptr addrspace(5) %loc
%r = add i32 %i, %j
%ext = zext i32 %r to i64
ret i64 %ext
@@ -17,7 +17,7 @@ define i64 @i32_test(i32 %i) nounwind readnone {
define i64 @i16_test(i16 %i) nounwind readnone {
%loc = alloca i16, addrspace(5)
- %j = load i16, ptr addrspace(5) %loc
+ %j = load volatile i16, ptr addrspace(5) %loc
%r = add i16 %i, %j
%ext = zext i16 %r to i64
ret i64 %ext
@@ -25,7 +25,7 @@ define i64 @i16_test(i16 %i) nounwind readnone {
define i64 @i8_test(i8 %i) nounwind readnone {
%loc = alloca i8, addrspace(5)
- %j = load i8, ptr addrspace(5) %loc
+ %j = load volatile i8, ptr addrspace(5) %loc
%r = add i8 %i, %j
%ext = zext i8 %r to i64
ret i64 %ext
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll.expected b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll.expected
index 6be3b24a735f3..fe380511a73a0 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll.expected
@@ -1,19 +1,19 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -disable-promote-alloca-to-vector < %s | FileCheck %s
+; RUN: llc -mtriple=amdgpu7.00-amd-amdhsa < %s | FileCheck %s
define i64 @i64_test(i64 %i) nounwind readnone {
; CHECK-LABEL: i64_test:
; CHECK: ; %bb.0:
; CHECK-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; CHECK-NEXT: buffer_load_dword v2, off, s[0:3], s32
-; CHECK-NEXT: buffer_load_dword v3, off, s[0:3], s32 offset:4
-; CHECK-NEXT: s_waitcnt vmcnt(1)
-; CHECK-NEXT: v_add_i32_e32 v0, vcc, v0, v2
+; CHECK-NEXT: buffer_load_dword v2, off, s[0:3], s32 glc
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: buffer_load_dword v3, off, s[0:3], s32 offset:4 glc
; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_add_i32_e32 v0, vcc, v0, v2
; CHECK-NEXT: v_addc_u32_e32 v1, vcc, v1, v3, vcc
; CHECK-NEXT: s_setpc_b64 s[30:31]
%loc = alloca i64, addrspace(5)
- %j = load i64, ptr addrspace(5) %loc
+ %j = load volatile i64, ptr addrspace(5) %loc
%r = add i64 %i, %j
ret i64 %r
}
@@ -22,13 +22,13 @@ define i64 @i32_test(i32 %i) nounwind readnone {
; CHECK-LABEL: i32_test:
; CHECK: ; %bb.0:
; CHECK-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; CHECK-NEXT: buffer_load_dword v1, off, s[0:3], s32
+; CHECK-NEXT: buffer_load_dword v1, off, s[0:3], s32 glc
; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: v_add_i32_e32 v0, vcc, v0, v1
; CHECK-NEXT: v_mov_b32_e32 v1, 0
; CHECK-NEXT: s_setpc_b64 s[30:31]
%loc = alloca i32, addrspace(5)
- %j = load i32, ptr addrspace(5) %loc
+ %j = load volatile i32, ptr addrspace(5) %loc
%r = add i32 %i, %j
%ext = zext i32 %r to i64
ret i64 %ext
@@ -38,14 +38,14 @@ define i64 @i16_test(i16 %i) nounwind readnone {
; CHECK-LABEL: i16_test:
; CHECK: ; %bb.0:
; CHECK-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; CHECK-NEXT: buffer_load_ushort v1, off, s[0:3], s32
+; CHECK-NEXT: buffer_load_ushort v1, off, s[0:3], s32 glc
; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: v_add_i32_e32 v0, vcc, v0, v1
; CHECK-NEXT: v_and_b32_e32 v0, 0xffff, v0
; CHECK-NEXT: v_mov_b32_e32 v1, 0
; CHECK-NEXT: s_setpc_b64 s[30:31]
%loc = alloca i16, addrspace(5)
- %j = load i16, ptr addrspace(5) %loc
+ %j = load volatile i16, ptr addrspace(5) %loc
%r = add i16 %i, %j
%ext = zext i16 %r to i64
ret i64 %ext
@@ -55,14 +55,14 @@ define i64 @i8_test(i8 %i) nounwind readnone {
; CHECK-LABEL: i8_test:
; CHECK: ; %bb.0:
; CHECK-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; CHECK-NEXT: buffer_load_ubyte v1, off, s[0:3], s32
+; CHECK-NEXT: buffer_load_ubyte v1, off, s[0:3], s32 glc
; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: v_add_i32_e32 v0, vcc, v0, v1
; CHECK-NEXT: v_and_b32_e32 v0, 0xff, v0
; CHECK-NEXT: v_mov_b32_e32 v1, 0
; CHECK-NEXT: s_setpc_b64 s[30:31]
%loc = alloca i8, addrspace(5)
- %j = load i8, ptr addrspace(5) %loc
+ %j = load volatile i8, ptr addrspace(5) %loc
%r = add i8 %i, %j
%ext = zext i8 %r to i64
ret i64 %ext
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll
index 42d2eb3984f10..083259eb10d12 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll
@@ -1,4 +1,4 @@
-; RUN: llc -enable-machine-outliner -disable-promote-alloca-to-vector -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
+; RUN: llc -enable-machine-outliner -mtriple=amdgpu7.00-amd-amdhsa < %s | FileCheck %s
; NOTE: Machine outliner doesn't run.
@x = dso_local global i32 0, align 4
@@ -9,32 +9,32 @@ define dso_local i32 @check_boundaries() #0 {
%3 = alloca i32, align 4, addrspace(5)
%4 = alloca i32, align 4, addrspace(5)
%5 = alloca i32, align 4, addrspace(5)
- store i32 0, i32 addrspace(5)* %1, align 4
- store i32 0, i32 addrspace(5)* %2, align 4
- %6 = load i32, i32 addrspace(5)* %2, align 4
+ store volatile i32 0, i32 addrspace(5)* %1, align 4
+ store volatile i32 0, i32 addrspace(5)* %2, align 4
+ %6 = load volatile i32, i32 addrspace(5)* %2, align 4
%7 = icmp ne i32 %6, 0
br i1 %7, label %9, label %8
- store i32 1, i32 addrspace(5)* %2, align 4
- store i32 2, i32 addrspace(5)* %3, align 4
- store i32 3, i32 addrspace(5)* %4, align 4
- store i32 4, i32 addrspace(5)* %5, align 4
+ store volatile i32 1, i32 addrspace(5)* %2, align 4
+ store volatile i32 2, i32 addrspace(5)* %3, align 4
+ store volatile i32 3, i32 addrspace(5)* %4, align 4
+ store volatile i32 4, i32 addrspace(5)* %5, align 4
br label %10
- store i32 1, i32 addrspace(5)* %4, align 4
+ store volatile i32 1, i32 addrspace(5)* %4, align 4
br label %10
- %11 = load i32, i32 addrspace(5)* %2, align 4
+ %11 = load volatile i32, i32 addrspace(5)* %2, align 4
%12 = icmp ne i32 %11, 0
br i1 %12, label %14, label %13
- store i32 1, i32 addrspace(5)* %2, align 4
- store i32 2, i32 addrspace(5)* %3, align 4
- store i32 3, i32 addrspace(5)* %4, align 4
- store i32 4, i32 addrspace(5)* %5, align 4
+ store volatile i32 1, i32 addrspace(5)* %2, align 4
+ store volatile i32 2, i32 addrspace(5)* %3, align 4
+ store volatile i32 3, i32 addrspace(5)* %4, align 4
+ store volatile i32 4, i32 addrspace(5)* %5, align 4
br label %15
- store i32 1, i32 addrspace(5)* %4, align 4
+ store volatile i32 1, i32 addrspace(5)* %4, align 4
br label %15
ret i32 0
@@ -47,18 +47,18 @@ define dso_local i32 @main() #0 {
%4 = alloca i32, align 4, addrspace(5)
%5 = alloca i32, align 4, addrspace(5)
- store i32 0, i32 addrspace(5)* %1, align 4
+ store volatile i32 0, i32 addrspace(5)* %1, align 4
store i32 0, i32* @x, align 4
- store i32 1, i32 addrspace(5)* %2, align 4
- store i32 2, i32 addrspace(5)* %3, align 4
- store i32 3, i32 addrspace(5)* %4, align 4
- store i32 4, i32 addrspace(5)* %5, align 4
+ store volatile i32 1, i32 addrspace(5)* %2, align 4
+ store volatile i32 2, i32 addrspace(5)* %3, align 4
+ store volatile i32 3, i32 addrspace(5)* %4, align 4
+ store volatile i32 4, i32 addrspace(5)* %5, align 4
store i32 1, i32* @x, align 4
call void asm sideeffect "", "~{memory},~{dirflag},~{fpsr},~{flags}"()
- store i32 1, i32 addrspace(5)* %2, align 4
- store i32 2, i32 addrspace(5)* %3, align 4
- store i32 3, i32 addrspace(5)* %4, align 4
- store i32 4, i32 addrspace(5)* %5, align 4
+ store volatile i32 1, i32 addrspace(5)* %2, align 4
+ store volatile i32 2, i32 addrspace(5)* %3, align 4
+ store volatile i32 3, i32 addrspace(5)* %4, align 4
+ store volatile i32 4, i32 addrspace(5)* %5, align 4
ret i32 0
}
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.generated.expected b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.generated.expected
index 02d207f5c8609..56c26ed262cd8 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.generated.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.generated.expected
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --include-generated-funcs
-; RUN: llc -enable-machine-outliner -disable-promote-alloca-to-vector -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
+; RUN: llc -enable-machine-outliner -mtriple=amdgpu7.00-amd-amdhsa < %s | FileCheck %s
; NOTE: Machine outliner doesn't run.
@x = dso_local global i32 0, align 4
@@ -10,32 +10,32 @@ define dso_local i32 @check_boundaries() #0 {
%3 = alloca i32, align 4, addrspace(5)
%4 = alloca i32, align 4, addrspace(5)
%5 = alloca i32, align 4, addrspace(5)
- store i32 0, i32 addrspace(5)* %1, align 4
- store i32 0, i32 addrspace(5)* %2, align 4
- %6 = load i32, i32 addrspace(5)* %2, align 4
+ store volatile i32 0, i32 addrspace(5)* %1, align 4
+ store volatile i32 0, i32 addrspace(5)* %2, align 4
+ %6 = load volatile i32, i32 addrspace(5)* %2, align 4
%7 = icmp ne i32 %6, 0
br i1 %7, label %9, label %8
- store i32 1, i32 addrspace(5)* %2, align 4
- store i32 2, i32 addrspace(5)* %3, align 4
- store i32 3, i32 addrspace(5)* %4, align 4
- store i32 4, i32 addrspace(5)* %5, align 4
+ store volatile i32 1, i32 addrspace(5)* %2, align 4
+ store volatile i32 2, i32 addrspace(5)* %3, align 4
+ store volatile i32 3, i32 addrspace(5)* %4, align 4
+ store volatile i32 4, i32 addrspace(5)* %5, align 4
br label %10
- store i32 1, i32 addrspace(5)* %4, align 4
+ store volatile i32 1, i32 addrspace(5)* %4, align 4
br label %10
- %11 = load i32, i32 addrspace(5)* %2, align 4
+ %11 = load volatile i32, i32 addrspace(5)* %2, align 4
%12 = icmp ne i32 %11, 0
br i1 %12, label %14, label %13
- store i32 1, i32 addrspace(5)* %2, align 4
- store i32 2, i32 addrspace(5)* %3, align 4
- store i32 3, i32 addrspace(5)* %4, align 4
- store i32 4, i32 addrspace(5)* %5, align 4
+ store volatile i32 1, i32 addrspace(5)* %2, align 4
+ store volatile i32 2, i32 addrspace(5)* %3, align 4
+ store volatile i32 3, i32 addrspace(5)* %4, align 4
+ store volatile i32 4, i32 addrspace(5)* %5, align 4
br label %15
- store i32 1, i32 addrspace(5)* %4, align 4
+ store volatile i32 1, i32 addrspace(5)* %4, align 4
br label %15
ret i32 0
@@ -48,18 +48,18 @@ define dso_local i32 @main() #0 {
%4 = alloca i32, align 4, addrspace(5)
%5 = alloca i32, align 4, addrspace(5)
- store i32 0, i32 addrspace(5)* %1, align 4
+ store volatile i32 0, i32 addrspace(5)* %1, align 4
store i32 0, i32* @x, align 4
- store i32 1, i32 addrspace(5)* %2, align 4
- store i32 2, i32 addrspace(5)* %3, align 4
- store i32 3, i32 addrspace(5)* %4, align 4
- store i32 4, i32 addrspace(5)* %5, align 4
+ store volatile i32 1, i32 addrspace(5)* %2, align 4
+ store volatile i32 2, i32 addrspace(5)* %3, align 4
+ store volatile i32 3, i32 addrspace(5)* %4, align 4
+ store volatile i32 4, i32 addrspace(5)* %5, align 4
store i32 1, i32* @x, align 4
call void asm sideeffect "", "~{memory},~{dirflag},~{fpsr},~{flags}"()
- store i32 1, i32 addrspace(5)* %2, align 4
- store i32 2, i32 addrspace(5)* %3, align 4
- store i32 3, i32 addrspace(5)* %4, align 4
- store i32 4, i32 addrspace(5)* %5, align 4
+ store volatile i32 1, i32 addrspace(5)* %2, align 4
+ store volatile i32 2, i32 addrspace(5)* %3, align 4
+ store volatile i32 3, i32 addrspace(5)* %4, align 4
+ store volatile i32 4, i32 addrspace(5)* %5, align 4
ret i32 0
}
@@ -72,52 +72,79 @@ attributes #0 = { noredzone nounwind ssp uwtable "frame-pointer"="all" }
; CHECK-NEXT: .cfi_llvm_def_aspace_cfa 64, 0, 6
; CHECK-NEXT: .cfi_llvm_register_pair 16, 62, 32, 63, 32
; CHECK-NEXT: .cfi_undefined 2560
-; CHECK-NEXT: .cfi_undefined 2561
-; CHECK-NEXT: .cfi_undefined 2562
-; CHECK-NEXT: .cfi_undefined 2563
-; CHECK-NEXT: .cfi_undefined 2564
; CHECK-NEXT: .cfi_undefined 36
; CHECK-NEXT: .cfi_undefined 37
-; CHECK-NEXT: .cfi_undefined 38
-; CHECK-NEXT: .cfi_undefined 39
; CHECK-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; CHECK-NEXT: s_mov_b32 s8, s33
-; CHECK-NEXT: .cfi_register 65, 40
+; CHECK-NEXT: s_mov_b32 s6, s33
+; CHECK-NEXT: .cfi_register 65, 38
; CHECK-NEXT: s_mov_b32 s33, s32
; CHECK-NEXT: .cfi_def_cfa_register 65
-; CHECK-NEXT: s_addk_i32 s32, 0x600
-; CHECK-NEXT: v_mov_b32_e32 v4, 0
-; CHECK-NEXT: v_mov_b32_e32 v0, 1
-; CHECK-NEXT: v_mov_b32_e32 v1, 2
-; CHECK-NEXT: v_mov_b32_e32 v2, 3
-; CHECK-NEXT: v_mov_b32_e32 v3, 4
-; CHECK-NEXT: buffer_store_dword v4, off, s[0:3], s33
+; CHECK-NEXT: v_mov_b32_e32 v0, 0
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:4
-; CHECK-NEXT: buffer_store_dword v1, off, s[0:3], s33 offset:8
-; CHECK-NEXT: buffer_store_dword v2, off, s[0:3], s33 offset:12
-; CHECK-NEXT: buffer_store_dword v3, off, s[0:3], s33 offset:16
-; CHECK-NEXT: s_mov_b64 s[4:5], 0
-; CHECK-NEXT: s_and_saveexec_b64 s[6:7], s[4:5]
-; CHECK-NEXT: s_xor_b64 s[4:5], exec, s[6:7]
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: buffer_load_dword v0, off, s[0:3], s33 offset:4 glc
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: s_addk_i32 s32, 0x600
+; CHECK-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
+; CHECK-NEXT: s_and_saveexec_b64 s[4:5], vcc
+; CHECK-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
; CHECK-NEXT: s_cbranch_execz .LBB0_2
; CHECK-NEXT: ; %bb.1:
+; CHECK-NEXT: v_mov_b32_e32 v0, 1
; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:4
-; CHECK-NEXT: buffer_store_dword v1, off, s[0:3], s33 offset:8
-; CHECK-NEXT: buffer_store_dword v2, off, s[0:3], s33 offset:12
-; CHECK-NEXT: buffer_store_dword v3, off, s[0:3], s33 offset:16
-; CHECK-NEXT: .LBB0_2: ; %Flow
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 2
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:8
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 3
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:12
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 4
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:16
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: .LBB0_2: ; %Flow1
; CHECK-NEXT: s_andn2_saveexec_b64 s[4:5], s[4:5]
; CHECK-NEXT: s_cbranch_execz .LBB0_4
; CHECK-NEXT: ; %bb.3:
; CHECK-NEXT: v_mov_b32_e32 v0, 1
; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:12
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: .LBB0_4:
; CHECK-NEXT: s_or_b64 exec, exec, s[4:5]
+; CHECK-NEXT: buffer_load_dword v0, off, s[0:3], s33 offset:4 glc
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
+; CHECK-NEXT: s_and_saveexec_b64 s[4:5], vcc
+; CHECK-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
+; CHECK-NEXT: s_cbranch_execz .LBB0_6
+; CHECK-NEXT: ; %bb.5:
+; CHECK-NEXT: v_mov_b32_e32 v0, 1
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:4
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 2
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:8
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 3
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:12
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 4
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:16
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: .LBB0_6: ; %Flow
+; CHECK-NEXT: s_andn2_saveexec_b64 s[4:5], s[4:5]
+; CHECK-NEXT: s_cbranch_execz .LBB0_8
+; CHECK-NEXT: ; %bb.7:
+; CHECK-NEXT: v_mov_b32_e32 v0, 1
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:12
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: .LBB0_8:
+; CHECK-NEXT: s_or_b64 exec, exec, s[4:5]
; CHECK-NEXT: v_mov_b32_e32 v0, 0
; CHECK-NEXT: s_mov_b32 s32, s33
; CHECK-NEXT: .cfi_def_cfa_register 64
-; CHECK-NEXT: s_mov_b32 s33, s8
-; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: s_mov_b32 s33, s6
; CHECK-NEXT: s_setpc_b64 s[30:31]
;
; CHECK-LABEL: main:
@@ -145,27 +172,36 @@ attributes #0 = { noredzone nounwind ssp uwtable "frame-pointer"="all" }
; CHECK-NEXT: s_getpc_b64 s[4:5]
; CHECK-NEXT: s_add_u32 s4, s4, x at rel32@lo+4
; CHECK-NEXT: s_addc_u32 s5, s5, x at rel32@hi+12
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: v_mov_b32_e32 v2, 1
; CHECK-NEXT: v_mov_b32_e32 v3, 2
; CHECK-NEXT: v_mov_b32_e32 v4, 3
; CHECK-NEXT: v_mov_b32_e32 v5, 4
-; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33
+; CHECK-NEXT: v_mov_b32_e32 v0, s4
+; CHECK-NEXT: v_mov_b32_e32 v1, s5
; CHECK-NEXT: buffer_store_dword v2, off, s[0:3], s33 offset:4
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v3, off, s[0:3], s33 offset:8
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v4, off, s[0:3], s33 offset:12
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v5, off, s[0:3], s33 offset:16
-; CHECK-NEXT: v_mov_b32_e32 v0, s4
-; CHECK-NEXT: v_mov_b32_e32 v1, s5
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: flat_store_dword v[0:1], v2
; CHECK-NEXT: ;;#ASMSTART
; CHECK-NEXT: ;;#ASMEND
; CHECK-NEXT: buffer_store_dword v2, off, s[0:3], s33 offset:4
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v3, off, s[0:3], s33 offset:8
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v4, off, s[0:3], s33 offset:12
-; CHECK-NEXT: v_mov_b32_e32 v0, 0
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v5, off, s[0:3], s33 offset:16
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 0
; CHECK-NEXT: s_mov_b32 s32, s33
; CHECK-NEXT: .cfi_def_cfa_register 64
; CHECK-NEXT: s_mov_b32 s33, s6
-; CHECK-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; CHECK-NEXT: s_waitcnt lgkmcnt(0)
; CHECK-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.nogenerated.expected b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.nogenerated.expected
index c913aeb16fbf8..ec26bf2481970 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.nogenerated.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.nogenerated.expected
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -enable-machine-outliner -disable-promote-alloca-to-vector -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
+; RUN: llc -enable-machine-outliner -mtriple=amdgpu7.00-amd-amdhsa < %s | FileCheck %s
; NOTE: Machine outliner doesn't run.
@x = dso_local global i32 0, align 4
@@ -13,84 +13,111 @@ define dso_local i32 @check_boundaries() #0 {
; CHECK-NEXT: .cfi_llvm_def_aspace_cfa 64, 0, 6
; CHECK-NEXT: .cfi_llvm_register_pair 16, 62, 32, 63, 32
; CHECK-NEXT: .cfi_undefined 2560
-; CHECK-NEXT: .cfi_undefined 2561
-; CHECK-NEXT: .cfi_undefined 2562
-; CHECK-NEXT: .cfi_undefined 2563
-; CHECK-NEXT: .cfi_undefined 2564
; CHECK-NEXT: .cfi_undefined 36
; CHECK-NEXT: .cfi_undefined 37
-; CHECK-NEXT: .cfi_undefined 38
-; CHECK-NEXT: .cfi_undefined 39
; CHECK-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; CHECK-NEXT: s_mov_b32 s8, s33
-; CHECK-NEXT: .cfi_register 65, 40
+; CHECK-NEXT: s_mov_b32 s6, s33
+; CHECK-NEXT: .cfi_register 65, 38
; CHECK-NEXT: s_mov_b32 s33, s32
; CHECK-NEXT: .cfi_def_cfa_register 65
-; CHECK-NEXT: s_addk_i32 s32, 0x600
-; CHECK-NEXT: v_mov_b32_e32 v4, 0
-; CHECK-NEXT: v_mov_b32_e32 v0, 1
-; CHECK-NEXT: v_mov_b32_e32 v1, 2
-; CHECK-NEXT: v_mov_b32_e32 v2, 3
-; CHECK-NEXT: v_mov_b32_e32 v3, 4
-; CHECK-NEXT: buffer_store_dword v4, off, s[0:3], s33
+; CHECK-NEXT: v_mov_b32_e32 v0, 0
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:4
-; CHECK-NEXT: buffer_store_dword v1, off, s[0:3], s33 offset:8
-; CHECK-NEXT: buffer_store_dword v2, off, s[0:3], s33 offset:12
-; CHECK-NEXT: buffer_store_dword v3, off, s[0:3], s33 offset:16
-; CHECK-NEXT: s_mov_b64 s[4:5], 0
-; CHECK-NEXT: s_and_saveexec_b64 s[6:7], s[4:5]
-; CHECK-NEXT: s_xor_b64 s[4:5], exec, s[6:7]
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: buffer_load_dword v0, off, s[0:3], s33 offset:4 glc
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: s_addk_i32 s32, 0x600
+; CHECK-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
+; CHECK-NEXT: s_and_saveexec_b64 s[4:5], vcc
+; CHECK-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
; CHECK-NEXT: s_cbranch_execz .LBB0_2
; CHECK-NEXT: ; %bb.1:
+; CHECK-NEXT: v_mov_b32_e32 v0, 1
; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:4
-; CHECK-NEXT: buffer_store_dword v1, off, s[0:3], s33 offset:8
-; CHECK-NEXT: buffer_store_dword v2, off, s[0:3], s33 offset:12
-; CHECK-NEXT: buffer_store_dword v3, off, s[0:3], s33 offset:16
-; CHECK-NEXT: .LBB0_2: ; %Flow
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 2
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:8
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 3
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:12
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 4
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:16
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: .LBB0_2: ; %Flow1
; CHECK-NEXT: s_andn2_saveexec_b64 s[4:5], s[4:5]
; CHECK-NEXT: s_cbranch_execz .LBB0_4
; CHECK-NEXT: ; %bb.3:
; CHECK-NEXT: v_mov_b32_e32 v0, 1
; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:12
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: .LBB0_4:
; CHECK-NEXT: s_or_b64 exec, exec, s[4:5]
+; CHECK-NEXT: buffer_load_dword v0, off, s[0:3], s33 offset:4 glc
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
+; CHECK-NEXT: s_and_saveexec_b64 s[4:5], vcc
+; CHECK-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
+; CHECK-NEXT: s_cbranch_execz .LBB0_6
+; CHECK-NEXT: ; %bb.5:
+; CHECK-NEXT: v_mov_b32_e32 v0, 1
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:4
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 2
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:8
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 3
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:12
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 4
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:16
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: .LBB0_6: ; %Flow
+; CHECK-NEXT: s_andn2_saveexec_b64 s[4:5], s[4:5]
+; CHECK-NEXT: s_cbranch_execz .LBB0_8
+; CHECK-NEXT: ; %bb.7:
+; CHECK-NEXT: v_mov_b32_e32 v0, 1
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33 offset:12
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: .LBB0_8:
+; CHECK-NEXT: s_or_b64 exec, exec, s[4:5]
; CHECK-NEXT: v_mov_b32_e32 v0, 0
; CHECK-NEXT: s_mov_b32 s32, s33
; CHECK-NEXT: .cfi_def_cfa_register 64
-; CHECK-NEXT: s_mov_b32 s33, s8
-; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: s_mov_b32 s33, s6
; CHECK-NEXT: s_setpc_b64 s[30:31]
%1 = alloca i32, align 4, addrspace(5)
%2 = alloca i32, align 4, addrspace(5)
%3 = alloca i32, align 4, addrspace(5)
%4 = alloca i32, align 4, addrspace(5)
%5 = alloca i32, align 4, addrspace(5)
- store i32 0, i32 addrspace(5)* %1, align 4
- store i32 0, i32 addrspace(5)* %2, align 4
- %6 = load i32, i32 addrspace(5)* %2, align 4
+ store volatile i32 0, i32 addrspace(5)* %1, align 4
+ store volatile i32 0, i32 addrspace(5)* %2, align 4
+ %6 = load volatile i32, i32 addrspace(5)* %2, align 4
%7 = icmp ne i32 %6, 0
br i1 %7, label %9, label %8
- store i32 1, i32 addrspace(5)* %2, align 4
- store i32 2, i32 addrspace(5)* %3, align 4
- store i32 3, i32 addrspace(5)* %4, align 4
- store i32 4, i32 addrspace(5)* %5, align 4
+ store volatile i32 1, i32 addrspace(5)* %2, align 4
+ store volatile i32 2, i32 addrspace(5)* %3, align 4
+ store volatile i32 3, i32 addrspace(5)* %4, align 4
+ store volatile i32 4, i32 addrspace(5)* %5, align 4
br label %10
- store i32 1, i32 addrspace(5)* %4, align 4
+ store volatile i32 1, i32 addrspace(5)* %4, align 4
br label %10
- %11 = load i32, i32 addrspace(5)* %2, align 4
+ %11 = load volatile i32, i32 addrspace(5)* %2, align 4
%12 = icmp ne i32 %11, 0
br i1 %12, label %14, label %13
- store i32 1, i32 addrspace(5)* %2, align 4
- store i32 2, i32 addrspace(5)* %3, align 4
- store i32 3, i32 addrspace(5)* %4, align 4
- store i32 4, i32 addrspace(5)* %5, align 4
+ store volatile i32 1, i32 addrspace(5)* %2, align 4
+ store volatile i32 2, i32 addrspace(5)* %3, align 4
+ store volatile i32 3, i32 addrspace(5)* %4, align 4
+ store volatile i32 4, i32 addrspace(5)* %5, align 4
br label %15
- store i32 1, i32 addrspace(5)* %4, align 4
+ store volatile i32 1, i32 addrspace(5)* %4, align 4
br label %15
ret i32 0
@@ -122,29 +149,38 @@ define dso_local i32 @main() #0 {
; CHECK-NEXT: s_getpc_b64 s[4:5]
; CHECK-NEXT: s_add_u32 s4, s4, x at rel32@lo+4
; CHECK-NEXT: s_addc_u32 s5, s5, x at rel32@hi+12
+; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: v_mov_b32_e32 v2, 1
; CHECK-NEXT: v_mov_b32_e32 v3, 2
; CHECK-NEXT: v_mov_b32_e32 v4, 3
; CHECK-NEXT: v_mov_b32_e32 v5, 4
-; CHECK-NEXT: buffer_store_dword v0, off, s[0:3], s33
+; CHECK-NEXT: v_mov_b32_e32 v0, s4
+; CHECK-NEXT: v_mov_b32_e32 v1, s5
; CHECK-NEXT: buffer_store_dword v2, off, s[0:3], s33 offset:4
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v3, off, s[0:3], s33 offset:8
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v4, off, s[0:3], s33 offset:12
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v5, off, s[0:3], s33 offset:16
-; CHECK-NEXT: v_mov_b32_e32 v0, s4
-; CHECK-NEXT: v_mov_b32_e32 v1, s5
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: flat_store_dword v[0:1], v2
; CHECK-NEXT: ;;#ASMSTART
; CHECK-NEXT: ;;#ASMEND
; CHECK-NEXT: buffer_store_dword v2, off, s[0:3], s33 offset:4
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v3, off, s[0:3], s33 offset:8
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v4, off, s[0:3], s33 offset:12
-; CHECK-NEXT: v_mov_b32_e32 v0, 0
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: buffer_store_dword v5, off, s[0:3], s33 offset:16
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, 0
; CHECK-NEXT: s_mov_b32 s32, s33
; CHECK-NEXT: .cfi_def_cfa_register 64
; CHECK-NEXT: s_mov_b32 s33, s6
-; CHECK-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; CHECK-NEXT: s_waitcnt lgkmcnt(0)
; CHECK-NEXT: s_setpc_b64 s[30:31]
%1 = alloca i32, align 4, addrspace(5)
%2 = alloca i32, align 4, addrspace(5)
@@ -152,18 +188,18 @@ define dso_local i32 @main() #0 {
%4 = alloca i32, align 4, addrspace(5)
%5 = alloca i32, align 4, addrspace(5)
- store i32 0, i32 addrspace(5)* %1, align 4
+ store volatile i32 0, i32 addrspace(5)* %1, align 4
store i32 0, i32* @x, align 4
- store i32 1, i32 addrspace(5)* %2, align 4
- store i32 2, i32 addrspace(5)* %3, align 4
- store i32 3, i32 addrspace(5)* %4, align 4
- store i32 4, i32 addrspace(5)* %5, align 4
+ store volatile i32 1, i32 addrspace(5)* %2, align 4
+ store volatile i32 2, i32 addrspace(5)* %3, align 4
+ store volatile i32 3, i32 addrspace(5)* %4, align 4
+ store volatile i32 4, i32 addrspace(5)* %5, align 4
store i32 1, i32* @x, align 4
call void asm sideeffect "", "~{memory},~{dirflag},~{fpsr},~{flags}"()
- store i32 1, i32 addrspace(5)* %2, align 4
- store i32 2, i32 addrspace(5)* %3, align 4
- store i32 3, i32 addrspace(5)* %4, align 4
- store i32 4, i32 addrspace(5)* %5, align 4
+ store volatile i32 1, i32 addrspace(5)* %2, align 4
+ store volatile i32 2, i32 addrspace(5)* %3, align 4
+ store volatile i32 3, i32 addrspace(5)* %4, align 4
+ store volatile i32 4, i32 addrspace(5)* %5, align 4
ret i32 0
}
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll
index 8f3e3e1f5eda9..fe3f24fe2413e 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll
@@ -1,15 +1,15 @@
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -disable-promote-alloca-to-vector -stop-after=finalize-isel -debug-only=isel -o /dev/null %s 2>&1 | FileCheck %s
+; RUN: llc -mtriple=amdgpu7.00-amd-amdhsa -stop-after=finalize-isel -debug-only=isel -o /dev/null %s 2>&1 | FileCheck %s
define i64 @i64_test(i64 %i) nounwind readnone {
%loc = alloca i64, addrspace(5)
- %j = load i64, ptr addrspace(5) %loc
+ %j = load volatile i64, ptr addrspace(5) %loc
%r = add i64 %i, %j
ret i64 %r
}
define i64 @i32_test(i32 %i) nounwind readnone {
%loc = alloca i32, addrspace(5)
- %j = load i32, ptr addrspace(5) %loc
+ %j = load volatile i32, ptr addrspace(5) %loc
%r = add i32 %i, %j
%ext = zext i32 %r to i64
ret i64 %ext
@@ -17,7 +17,7 @@ define i64 @i32_test(i32 %i) nounwind readnone {
define i64 @i16_test(i16 %i) nounwind readnone {
%loc = alloca i16, addrspace(5)
- %j = load i16, ptr addrspace(5) %loc
+ %j = load volatile i16, ptr addrspace(5) %loc
%r = add i16 %i, %j
%ext = zext i16 %r to i64
ret i64 %ext
@@ -25,7 +25,7 @@ define i64 @i16_test(i16 %i) nounwind readnone {
define i64 @i8_test(i8 %i) nounwind readnone {
%loc = alloca i8, addrspace(5)
- %j = load i8, ptr addrspace(5) %loc
+ %j = load volatile i8, ptr addrspace(5) %loc
%r = add i8 %i, %j
%ext = zext i8 %r to i64
ret i64 %ext
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll.expected b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll.expected
index ada56f9b08d12..2131555c63f69 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll.expected
@@ -1,25 +1,26 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -disable-promote-alloca-to-vector -stop-after=finalize-isel -debug-only=isel -o /dev/null %s 2>&1 | FileCheck %s
+; RUN: llc -mtriple=amdgpu7.00-amd-amdhsa -stop-after=finalize-isel -debug-only=isel -o /dev/null %s 2>&1 | FileCheck %s
define i64 @i64_test(i64 %i) nounwind readnone {
; CHECK-LABEL: i64_test:
-; CHECK: SelectionDAG has 26 nodes:
+; CHECK: SelectionDAG has 27 nodes:
; CHECK-NEXT: t0: ch,glue = EntryToken
+; CHECK-NEXT: t27: i32,ch = BUFFER_LOAD_DWORD_OFFEN<Mem:(volatile dereferenceable load (s32) from %ir.loc, align 8, addrspace 5)> TargetFrameIndex:i32<0>, Register:v4i32 $sgpr0_sgpr1_sgpr2_sgpr3, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i1<0>, t0
+; CHECK-NEXT: t30: i32,ch = BUFFER_LOAD_DWORD_OFFEN<Mem:(volatile dereferenceable load (s32) from %ir.loc + 4, basealign 8, addrspace 5)> TargetFrameIndex:i32<0>, Register:v4i32 $sgpr0_sgpr1_sgpr2_sgpr3, TargetConstant:i32<0>, TargetConstant:i32<4>, TargetConstant:i32<0>, TargetConstant:i1<0>, t0
; CHECK-NEXT: t2: i32,ch = CopyFromReg # D:1 t0, Register:i32 %8
; CHECK-NEXT: t4: i32,ch = CopyFromReg # D:1 t0, Register:i32 %9
; CHECK-NEXT: t51: i64 = REG_SEQUENCE # D:1 TargetConstant:i32<67>, t2, TargetConstant:i32<3>, t4, TargetConstant:i32<11>
-; CHECK-NEXT: t27: i32,ch = BUFFER_LOAD_DWORD_OFFEN<Mem:(dereferenceable load (s32) from %ir.loc, align 8, addrspace 5)> TargetFrameIndex:i32<0>, Register:v4i32 $sgpr0_sgpr1_sgpr2_sgpr3, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i1<0>, t0
-; CHECK-NEXT: t30: i32,ch = BUFFER_LOAD_DWORD_OFFEN<Mem:(dereferenceable load (s32) from %ir.loc + 4, basealign 8, addrspace 5)> TargetFrameIndex:i32<0>, Register:v4i32 $sgpr0_sgpr1_sgpr2_sgpr3, TargetConstant:i32<0>, TargetConstant:i32<4>, TargetConstant:i32<0>, TargetConstant:i1<0>, t0
; CHECK-NEXT: t33: v2i32 = REG_SEQUENCE # D:1 TargetConstant:i32<37>, t27, TargetConstant:i32<3>, t30, TargetConstant:i32<11>
; CHECK-NEXT: t10: i64 = V_ADD_U64_PSEUDO # D:1 t51, t33
+; CHECK-NEXT: t32: ch = TokenFactor t27:1, t30:1
; CHECK-NEXT: t24: i32 = EXTRACT_SUBREG # D:1 t10, TargetConstant:i32<3>
-; CHECK-NEXT: t17: ch,glue = CopyToReg # D:1 t0, Register:i32 $vgpr0, t24
+; CHECK-NEXT: t17: ch,glue = CopyToReg # D:1 t32, Register:i32 $vgpr0, t24
; CHECK-NEXT: t39: i32 = EXTRACT_SUBREG # D:1 t10, TargetConstant:i32<11>
; CHECK-NEXT: t19: ch,glue = CopyToReg # D:1 t17, Register:i32 $vgpr1, t39, t17:1
; CHECK-NEXT: t20: ch = SI_RETURN Register:i32 $vgpr0, Register:i32 $vgpr1, t19, t19:1
; CHECK-EMPTY:
%loc = alloca i64, addrspace(5)
- %j = load i64, ptr addrspace(5) %loc
+ %j = load volatile i64, ptr addrspace(5) %loc
%r = add i64 %i, %j
ret i64 %r
}
@@ -28,16 +29,16 @@ define i64 @i32_test(i32 %i) nounwind readnone {
; CHECK-LABEL: i32_test:
; CHECK: SelectionDAG has 15 nodes:
; CHECK-NEXT: t0: ch,glue = EntryToken
+; CHECK-NEXT: t6: i32,ch = BUFFER_LOAD_DWORD_OFFEN<Mem:(volatile dereferenceable load (s32) from %ir.loc, addrspace 5)> TargetFrameIndex:i32<0>, Register:v4i32 $sgpr0_sgpr1_sgpr2_sgpr3, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i1<0>, t0
; CHECK-NEXT: t2: i32,ch = CopyFromReg # D:1 t0, Register:i32 %8
-; CHECK-NEXT: t6: i32,ch = BUFFER_LOAD_DWORD_OFFEN<Mem:(dereferenceable load (s32) from %ir.loc, addrspace 5)> TargetFrameIndex:i32<0>, Register:v4i32 $sgpr0_sgpr1_sgpr2_sgpr3, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i1<0>, t0
; CHECK-NEXT: t7: i32,i1 = V_ADD_CO_U32_e64 # D:1 t2, t6, TargetConstant:i1<0>
-; CHECK-NEXT: t15: ch,glue = CopyToReg # D:1 t0, Register:i32 $vgpr0, t7
+; CHECK-NEXT: t15: ch,glue = CopyToReg # D:1 t6:1, Register:i32 $vgpr0, t7
; CHECK-NEXT: t24: i32 = V_MOV_B32_e32 TargetConstant:i32<0>
; CHECK-NEXT: t17: ch,glue = CopyToReg t15, Register:i32 $vgpr1, t24, t15:1
; CHECK-NEXT: t18: ch = SI_RETURN Register:i32 $vgpr0, Register:i32 $vgpr1, t17, t17:1
; CHECK-EMPTY:
%loc = alloca i32, addrspace(5)
- %j = load i32, ptr addrspace(5) %loc
+ %j = load volatile i32, ptr addrspace(5) %loc
%r = add i32 %i, %j
%ext = zext i32 %r to i64
ret i64 %ext
@@ -47,18 +48,18 @@ define i64 @i16_test(i16 %i) nounwind readnone {
; CHECK-LABEL: i16_test:
; CHECK: SelectionDAG has 18 nodes:
; CHECK-NEXT: t0: ch,glue = EntryToken
+; CHECK-NEXT: t20: i32,ch = BUFFER_LOAD_USHORT_OFFEN<Mem:(volatile dereferenceable load (s16) from %ir.loc, addrspace 5)> TargetFrameIndex:i32<0>, Register:v4i32 $sgpr0_sgpr1_sgpr2_sgpr3, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i1<0>, t0
; CHECK-NEXT: t2: i32,ch = CopyFromReg # D:1 t0, Register:i32 %8
-; CHECK-NEXT: t20: i32,ch = BUFFER_LOAD_USHORT_OFFEN<Mem:(dereferenceable load (s16) from %ir.loc, addrspace 5)> TargetFrameIndex:i32<0>, Register:v4i32 $sgpr0_sgpr1_sgpr2_sgpr3, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i1<0>, t0
; CHECK-NEXT: t21: i32,i1 = V_ADD_CO_U32_e64 # D:1 t2, t20, TargetConstant:i1<0>
; CHECK-NEXT: t25: i32 = S_MOV_B32 TargetConstant:i32<65535>
; CHECK-NEXT: t26: i32 = V_AND_B32_e64 # D:1 t21, t25
-; CHECK-NEXT: t16: ch,glue = CopyToReg # D:1 t0, Register:i32 $vgpr0, t26
+; CHECK-NEXT: t16: ch,glue = CopyToReg # D:1 t20:1, Register:i32 $vgpr0, t26
; CHECK-NEXT: t33: i32 = V_MOV_B32_e32 TargetConstant:i32<0>
; CHECK-NEXT: t18: ch,glue = CopyToReg t16, Register:i32 $vgpr1, t33, t16:1
; CHECK-NEXT: t19: ch = SI_RETURN Register:i32 $vgpr0, Register:i32 $vgpr1, t18, t18:1
; CHECK-EMPTY:
%loc = alloca i16, addrspace(5)
- %j = load i16, ptr addrspace(5) %loc
+ %j = load volatile i16, ptr addrspace(5) %loc
%r = add i16 %i, %j
%ext = zext i16 %r to i64
ret i64 %ext
@@ -68,18 +69,18 @@ define i64 @i8_test(i8 %i) nounwind readnone {
; CHECK-LABEL: i8_test:
; CHECK: SelectionDAG has 18 nodes:
; CHECK-NEXT: t0: ch,glue = EntryToken
+; CHECK-NEXT: t20: i32,ch = BUFFER_LOAD_UBYTE_OFFEN<Mem:(volatile dereferenceable load (s8) from %ir.loc, addrspace 5)> TargetFrameIndex:i32<0>, Register:v4i32 $sgpr0_sgpr1_sgpr2_sgpr3, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i1<0>, t0
; CHECK-NEXT: t2: i32,ch = CopyFromReg # D:1 t0, Register:i32 %8
-; CHECK-NEXT: t20: i32,ch = BUFFER_LOAD_UBYTE_OFFEN<Mem:(dereferenceable load (s8) from %ir.loc, addrspace 5)> TargetFrameIndex:i32<0>, Register:v4i32 $sgpr0_sgpr1_sgpr2_sgpr3, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i32<0>, TargetConstant:i1<0>, t0
; CHECK-NEXT: t21: i32,i1 = V_ADD_CO_U32_e64 # D:1 t2, t20, TargetConstant:i1<0>
; CHECK-NEXT: t25: i32 = S_MOV_B32 TargetConstant:i32<255>
; CHECK-NEXT: t26: i32 = V_AND_B32_e64 # D:1 t21, t25
-; CHECK-NEXT: t16: ch,glue = CopyToReg # D:1 t0, Register:i32 $vgpr0, t26
+; CHECK-NEXT: t16: ch,glue = CopyToReg # D:1 t20:1, Register:i32 $vgpr0, t26
; CHECK-NEXT: t33: i32 = V_MOV_B32_e32 TargetConstant:i32<0>
; CHECK-NEXT: t18: ch,glue = CopyToReg t16, Register:i32 $vgpr1, t33, t16:1
; CHECK-NEXT: t19: ch = SI_RETURN Register:i32 $vgpr0, Register:i32 $vgpr1, t18, t18:1
; CHECK-EMPTY:
%loc = alloca i8, addrspace(5)
- %j = load i8, ptr addrspace(5) %loc
+ %j = load volatile i8, ptr addrspace(5) %loc
%r = add i8 %i, %j
%ext = zext i8 %r to i64
ret i64 %ext
>From 4f1e5545f65ec940d05cdf7c6ef9c8e7bb8d83fb Mon Sep 17 00:00:00 2001
From: Ruiling Song <ruiling.song at amd.com>
Date: Fri, 18 Sep 2026 10:24:17 +0800
Subject: [PATCH 7/9] test: update the test after merge
The s_waitcnt was moved up by my change to make the load volatile.
---
llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll | 1 -
1 file changed, 1 deletion(-)
diff --git a/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll b/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll
index 47dfb192137a4..80d434fec27c5 100644
--- a/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll
+++ b/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll
@@ -165,7 +165,6 @@ define void @sgpr_scavenge_fi_stack_id(double %input, i1 %enter_fma_path, i1 %re
; CHECK-NEXT: v_mov_b32_e32 v7, vcc_lo
; CHECK-NEXT: s_mov_b32 vcc_lo, 1
; CHECK-NEXT: s_mul_i32 s32, s32, 64
-; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: v_mul_f64 v[8:9], v[4:5], v[8:9]
; CHECK-NEXT: buffer_store_dword v9, off, s[0:3], 0 offset:4
; CHECK-NEXT: buffer_store_dword v8, off, s[0:3], 0
>From 5f1ba4e3ff3d4d89e56d359feaa43a31a41b1a85 Mon Sep 17 00:00:00 2001
From: Ruiling Song <ruiling.song at amd.com>
Date: Mon, 21 Sep 2026 16:23:13 +0800
Subject: [PATCH 8/9] AMDGPU: Rewrite alloca group formation
First form the group based on EquivalenceClasses.
Then propagate the Uses etc to the AllocaAnalysis which is represented
by leader alloca.
This also change `Uses` back to ordered SetVector. This helps ensuring
deterministic visiting order of uses. Different visiting order
may cause different IR. Although they are all correct. Using ordered set
helps making the test promote-alloca-proper-value-replacement.ll always
get the same output.
---
.../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 135 +++++++++---------
...promote-alloca-proper-value-replacement.ll | 3 +-
2 files changed, 69 insertions(+), 69 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index f80998965fe8e..6dc39875bf3d7 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -28,6 +28,7 @@
#include "AMDGPU.h"
#include "GCNSubtarget.h"
#include "Utils/AMDGPUBaseInfo.h"
+#include "llvm/ADT/EquivalenceClasses.h"
#include "llvm/ADT/STLExtras.h"
#include "llvm/Analysis/CaptureTracking.h"
#include "llvm/Analysis/InstSimplifyFolder.h"
@@ -105,24 +106,33 @@ struct MemTransferInfo {
ConstantInt *DestIndex = nullptr;
};
+// Per-alloca record produced while scanning a single alloca's users
+struct AllocaUsesInfo {
+ AllocaInst *Alloca;
+ SmallSetVector<Use *, 8> Uses;
+ // Other allocas this alloca's pointer is merged with via a phi/select.
+ SmallVector<AllocaInst *> Links;
+ // True if some phi/select merges this alloca's pointer with null or a
+ // non-alloca object.
+ bool HaveUnpromotableMerge = false;
+
+ explicit AllocaUsesInfo(AllocaInst *Alloca) : Alloca(Alloca) {}
+};
+
// Analysis for planning the different strategies of alloca promotion.
//
// An AllocaAnalysis may represent one alloca or a group of allocas if they need
// to be promoted together.
struct AllocaAnalysis {
- DenseSet<Value *> Pointers;
- SmallDenseSet<Use *> Uses;
+ SmallSetVector<Use *, 8> Uses;
unsigned Score = 0;
- // True if some phi/select merges this alloca's pointer with null or a
- // non-alloca object. Such merges are still fine for LDS promotion, but they
- // disable vector promotion.
+ // True if any member's pointer is merged by a phi/select with null or a
+ // non-alloca object. Disables vector promotion for the whole group.
bool HaveUnpromotableMerge = false;
- // All member allocas of this group, in lane order (Members[0] == Alloca).
+ // All member allocas of this group, in lane order (Members[0] == leader,
+ // the earliest alloca in program order).
SmallVector<AllocaInst *> Members;
- // Other allocas this group is directly linked to via a phi/select of
- // pointers. Used to form groups; empty once grouping is complete.
- SmallVector<AllocaInst *> Links;
struct {
FixedVectorType *Ty = nullptr;
@@ -141,11 +151,9 @@ struct AllocaAnalysis {
SmallVector<User *> Worklist;
} LDS;
- explicit AllocaAnalysis(AllocaInst *Alloca) { Members.push_back(Alloca); }
-
bool isGroup() const { return Members.size() > 1; }
- // Returns the leader alloca if this is an alloca group.
+ // Returns the leader alloca of the group.
AllocaInst *getLeaderAlloca() const { return Members[0]; }
};
@@ -170,7 +178,7 @@ class AMDGPUPromoteAllocaImpl {
std::pair<Value *, Value *> getLocalSizeYZ(IRBuilder<> &Builder);
Value *getWorkitemID(IRBuilder<> &Builder, unsigned N);
- bool collectAllocaUses(AllocaAnalysis &AA) const;
+ bool collectAllocaUses(AllocaUsesInfo &Info) const;
/// Val is a derived pointer from Alloca. OpIdx0/OpIdx1 are the operand
/// indices to an instruction with 2 pointer inputs (e.g. select, icmp).
@@ -297,14 +305,14 @@ FunctionPass *llvm::createAMDGPUPromoteAlloca() {
return new AMDGPUPromoteAlloca();
}
-bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
+bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaUsesInfo &Info) const {
const auto RejectUser = [&](Instruction *Inst, Twine Msg) {
LLVM_DEBUG(dbgs() << " Cannot promote alloca: " << Msg << "\n"
<< " " << *Inst << "\n");
return false;
};
- SmallVector<Instruction *, 4> WorkList({AA.getLeaderAlloca()});
+ SmallVector<Instruction *, 4> WorkList({Info.Alloca});
SmallPtrSet<Value *, 8> Visited;
// Other allocas which are merged by phi/select with current alloca.
@@ -321,7 +329,7 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
return RejectUser(Inst, "pointer escapes via store");
}
}
- AA.Uses.insert(&U);
+ Info.Uses.insert(&U);
if (isa<GetElementPtrInst>(U.getUser())) {
WorkList.push_back(Inst);
@@ -331,12 +339,12 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
for (auto *Obj : BaseObjs) {
if (auto *BaseAlloca = dyn_cast<AllocaInst>(Obj)) {
- if (BaseAlloca != AA.getLeaderAlloca()) {
+ if (BaseAlloca != Info.Alloca) {
MergedAllocas.insert(const_cast<AllocaInst *>(BaseAlloca));
}
WorkList.push_back(Inst);
} else if (isa<ConstantPointerNull, ConstantAggregateZero>(Obj)) {
- AA.HaveUnpromotableMerge = true;
+ Info.HaveUnpromotableMerge = true;
WorkList.push_back(Inst);
} else {
return RejectUser(Inst, "phi/select with unkown object");
@@ -346,7 +354,7 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
}
}
for (auto *Alloca : MergedAllocas)
- AA.Links.push_back(Alloca);
+ Info.Links.push_back(Alloca);
return true;
}
@@ -398,7 +406,7 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
: (MaxVGPRs * 32)) /
VGPRBudgetRatio;
- SmallMapVector<AllocaInst *, AllocaAnalysis, 4> AllocaAnalysisMap;
+ SmallMapVector<AllocaInst *, AllocaUsesInfo, 4> PromotableAllocas;
for (Instruction &I : F.getEntryBlock()) {
auto *AI = dyn_cast<AllocaInst>(&I);
// Array allocations are probably not worth handling, since an allocation
@@ -406,65 +414,56 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
if (!AI || !AI->isStaticAlloca() || AI->isArrayAllocation())
continue;
LLVM_DEBUG(dbgs() << "Analyzing: " << *AI << '\n');
- AllocaAnalysis AA{AI};
- if (!collectAllocaUses(AA))
+ AllocaUsesInfo Info{AI};
+ if (!collectAllocaUses(Info))
continue;
- AllocaAnalysisMap.insert({AI, std::move(AA)});
+ PromotableAllocas.insert({AI, std::move(Info)});
+ }
+
+ // Group allocas that are merged by phi/select using a union-find over the
+ // link edges.
+ EquivalenceClasses<AllocaInst *> Groups;
+ for (auto &[AI, Info] : PromotableAllocas) {
+ Groups.insert(AI);
+ for (AllocaInst *Link : Info.Links)
+ Groups.unionSets(AI, Link);
}
std::vector<AllocaAnalysis> Allocas;
- for (auto &AA : AllocaAnalysisMap.values()) {
- if (AA.Uses.empty())
+ // Construct one AllocaAnalysis for each alloca group and merge the
+ // information from each alloca.
+ for (const auto *ECV : Groups) {
+ if (!ECV->isLeader())
continue;
- bool Promotable = true;
- SmallVector<AllocaInst *> Worklist;
- LLVM_DEBUG(dbgs() << "Process leader alloca " << *AA.getLeaderAlloca()
- << '\n');
- Worklist.append(AA.Links);
- // Pull the information from links into the leader alloca.
- while (!Worklist.empty()) {
- AllocaInst *CurLink = Worklist.pop_back_val();
- auto LinkIter = AllocaAnalysisMap.find(CurLink);
- if (LinkIter != AllocaAnalysisMap.end()) {
- AllocaAnalysis &LinkAA = LinkIter->second;
- // Skip if the linked alloca was done or points to the leader alloca
- // itself.
- if (LinkAA.Uses.empty() || &LinkAA == &AA)
- continue;
-
- LLVM_DEBUG({
- dbgs() << " Process link: " << *LinkAA.getLeaderAlloca() << '\n';
- for (auto *X : LinkAA.Uses)
- dbgs() << " Add User " << *X->getUser() << '\n';
- });
-
- AA.Uses.insert_range(LinkAA.Uses);
- LinkAA.Uses.clear();
- AA.HaveUnpromotableMerge |= LinkAA.HaveUnpromotableMerge;
- AA.Members.push_back(LinkAA.getLeaderAlloca());
-
- // Add indirect links to worklist, so that their info are properly
- // propagated to the leader alloca.
- Worklist.append(LinkAA.Links);
- LinkAA.Links.clear();
- } else {
- LLVM_DEBUG(dbgs() << " The alloca is not promotable\n");
- Promotable = false;
+ bool HasNonCandidateMember = false;
+ AllocaAnalysis AA;
+
+ for (AllocaInst *AI : Groups.members(*ECV)) {
+ auto It = PromotableAllocas.find(AI);
+ // A member that is not a promotion candidate makes the whole class
+ // unpromotable as a combined vector.
+ if (It == PromotableAllocas.end()) {
+ HasNonCandidateMember = true;
break;
}
+ AllocaUsesInfo &Info = It->second;
+ AA.Members.push_back(AI);
+ AA.Uses.insert_range(Info.Uses);
+ AA.HaveUnpromotableMerge |= Info.HaveUnpromotableMerge;
}
- if (Promotable) {
- sort(AA.Members, [](AllocaInst *A, AllocaInst *B) -> bool {
- return A->comesBefore(B);
- });
- LLVM_DEBUG({
- for (AllocaInst *M : AA.Members)
- dbgs() << " Members: " << *M << '\n';
- });
- Allocas.push_back(AA);
+ if (HasNonCandidateMember) {
+ LLVM_DEBUG(dbgs() << " Group containing " << *ECV->getData()
+ << " is not promotable\n");
+ continue;
}
+
+ sort(AA.Members, [](AllocaInst *A, AllocaInst *B) -> bool {
+ return A->comesBefore(B);
+ });
+
+ Allocas.push_back(std::move(AA));
}
for (AllocaAnalysis &AA : Allocas) {
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-proper-value-replacement.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-proper-value-replacement.ll
index befd06ea429b3..de150a149d4bb 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-proper-value-replacement.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-proper-value-replacement.ll
@@ -41,9 +41,10 @@ define half @forwarded_load_across_blocks() {
; CHECK-NEXT: [[ARR:%.*]] = freeze <4 x half> poison
; CHECK-NEXT: br label %[[BB2:.*]]
; CHECK: [[BB2]]:
+; CHECK-NEXT: [[TMP0:%.*]] = freeze <4 x half> <half 1.000000e+00, half 2.000000e+00, half 3.000000e+00, half 4.000000e+00>
; CHECK-NEXT: br label %[[BB3:.*]]
; CHECK: [[BB3]]:
-; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x half> <half 1.000000e+00, half 2.000000e+00, half 3.000000e+00, half 4.000000e+00>, i32 0
+; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x half> [[TMP0]], i32 0
; CHECK-NEXT: ret half [[TMP1]]
;
entry:
>From 5314c3ad1f5dc75e2a2dc0c377058fca8ceda7ae Mon Sep 17 00:00:00 2001
From: Ruiling Song <ruiling.song at amd.com>
Date: Tue, 22 Sep 2026 08:45:16 +0800
Subject: [PATCH 9/9] AMDGPU: hoist IRBuilder to be reusable
---
llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 10 ++++------
1 file changed, 4 insertions(+), 6 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index 6dc39875bf3d7..0cb48b27aefaa 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -549,7 +549,7 @@ static bool isSupportedMemset(MemSetInst *I, AllocaInst *AI,
}
static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
- LLVMContext &Ctx = Ptr->getContext();
+ IRBuilder<> B(Ptr->getContext());
Ptr = Ptr->stripPointerCasts();
// Already computed (also breaks cycles for pointer phis).
@@ -560,8 +560,7 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
// A pointer that is directly one of the member allocas indexes lane
// BaseLane[member].
if (auto *AI = dyn_cast<AllocaInst>(Ptr)) {
- Value *Idx =
- ConstantInt::get(Type::getInt32Ty(Ctx), AA.Vector.BaseLane.lookup(AI));
+ Value *Idx = B.getInt32(AA.Vector.BaseLane.lookup(AI));
AA.Vector.IndexCache[Ptr] = Idx;
return Idx;
}
@@ -569,7 +568,7 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
// Pointer phi: build a parallel phi of lane indices. Insert and cache it
// before recursing so self-referential phis terminate.
if (auto *Phi = dyn_cast<PHINode>(Ptr)) {
- IRBuilder<> B(Phi);
+ B.SetInsertPoint(Phi);
PHINode *IdxPhi = B.CreatePHI(B.getInt32Ty(), Phi->getNumIncomingValues(),
"promotealloca.idx");
AA.Vector.IndexCache[Ptr] = IdxPhi;
@@ -584,7 +583,7 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
if (auto *SI = dyn_cast<SelectInst>(Ptr)) {
Value *T = calculateVectorIndex(SI->getTrueValue(), AA);
Value *F = calculateVectorIndex(SI->getFalseValue(), AA);
- IRBuilder<> B(SI);
+ B.SetInsertPoint(SI);
Value *Idx = B.CreateSelect(SI->getCondition(), T, F, "promotealloca.idx");
AA.Vector.IndexCache[Ptr] = Idx;
return Idx;
@@ -595,7 +594,6 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
assert(I != AA.Vector.GEPVectorIdx.end() && "Must have entry for GEP!");
if (!I->second.Full) {
- IRBuilder<> B(Ctx);
Value *Result = nullptr;
B.SetInsertPoint(GEP);
More information about the llvm-commits
mailing list