[llvm] [AMDGPU][DRAFT] PromoteAllocaToVector - Promo alloca phi (PR #226353)
Yoonseo Choi via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 29 12:57:39 PDT 2026
https://github.com/yoonseoch updated https://github.com/llvm/llvm-project/pull/226353
>From 482fc054202392d5dec921d2daf3f34ae544ef5a Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 22 Sep 2026 01:27:03 +0000
Subject: [PATCH 1/6] Testing relaxing phi with more than 2 incoming values
---
llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 18 +++++++++++++++++-
1 file changed, 17 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index aefcabd0d3bf17..f40e918d718e56 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -155,6 +155,9 @@ class AMDGPUPromoteAllocaImpl {
Instruction *UseInst, int OpIdx0,
int OpIdx1) const;
+ bool allOpsAreDerivedFromSameAlloca(Value *Alloca, Value *Val,
+ Instruction *Inst) const;
+
/// Check whether we have enough local memory for promotion.
bool hasSufficientLocalMem(const Function &F);
@@ -316,7 +319,8 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
return RejectUser(Inst, "phi from mixed objects");
break;
default:
- return RejectUser(Inst, "phi with too many operands");
+ if (!allOpsAreDerivedFromSameAlloca(AA.Alloca, Cur, Phi))
+ return RejectUser(Inst, "phi with too many operands");
}
WorkList.push_back(Inst);
@@ -1380,6 +1384,18 @@ bool AMDGPUPromoteAllocaImpl::binaryOpIsDerivedFromSameAlloca(
return true;
}
+bool AMDGPUPromoteAllocaImpl::allOpsAreDerivedFromSameAlloca(
+ Value *Alloca, Value *Val, Instruction *Inst) const {
+ for (int i = 0; i < Inst->getNumOperands(); i++) {
+ Value *Op = Inst->getOperand(i);
+ if (Op == Val)
+ continue;
+ if (!binaryOpIsDerivedFromSameAlloca(Alloca, Op, Inst, i, i))
+ return false;
+ }
+ return true;
+}
+
void AMDGPUPromoteAllocaImpl::analyzePromoteToLDS(AllocaAnalysis &AA) const {
if (DisablePromoteAllocaToLDS) {
LLVM_DEBUG(dbgs() << " Promote alloca to LDS is disabled\n");
>From f240f382f405d2ea7df0d63fe796038bb8296596 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Thu, 24 Sep 2026 06:42:33 +0000
Subject: [PATCH 2/6] Extend promote alloca to vector for phi/select
---
.../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 208 +++++++++++++-----
.../AMDGPU/promote-alloca-to-lds-select.ll | 5 +-
2 files changed, 156 insertions(+), 57 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index f40e918d718e56..711f90e0762c49 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -29,6 +29,7 @@
#include "GCNSubtarget.h"
#include "Utils/AMDGPUBaseInfo.h"
#include "llvm/ADT/STLExtras.h"
+#include "llvm/ADT/SetVector.h"
#include "llvm/Analysis/CaptureTracking.h"
#include "llvm/Analysis/InstSimplifyFolder.h"
#include "llvm/Analysis/InstructionSimplify.h"
@@ -86,14 +87,20 @@ static cl::opt<unsigned>
"when sorting profitable allocas"),
cl::init(4));
-// We support vector indices of the form ((A * stride) >> shift) + B
-// VarIndex is A, VarMul is stride, VarShift is shift and ConstIndex is B. All
-// parts are optional.
-struct GEPToVectorIndex {
+// We support vector indices of the form ((A * stride) >> shift) + B + index(P)
+// VarIndex is A, VarMul is stride, VarShift is shift, ConstIndex is B and Base
+// is P. All parts are optional.
+//
+// Base is set when the pointer is derived from a phi or select instead of
+// directly from the alloca, in which case its index is computed recursively.
+// Because there is only one variable slot, a pointer with a Base must have no
+// variable offset of its own.
+struct PtrToVectorIndex {
WeakTrackingVH VarIndex = nullptr; // defaults to 0
ConstantInt *VarMul = nullptr; // defaults to 1
ConstantInt *VarShift = nullptr; // defaults to 0
ConstantInt *ConstIndex = nullptr; // defaults to 0
+ Value *Base = nullptr; // defaults to the alloca (index 0)
Value *Full = nullptr;
};
@@ -108,12 +115,14 @@ struct AllocaAnalysis {
DenseSet<Value *> Pointers;
SmallVector<Use *> Uses;
unsigned Score = 0;
- bool HaveSelectOrPHI = false;
struct {
FixedVectorType *Ty = nullptr;
SmallVector<Instruction *> Worklist;
- SmallVector<Instruction *> UsersToRemove;
- MapVector<GetElementPtrInst *, GEPToVectorIndex> GEPVectorIdx;
+ // A pointer may be reached through more than one use edge (e.g. a phi whose
+ // incoming values are all derived from the alloca), so this must not hold
+ // duplicates.
+ SetVector<Instruction *> UsersToRemove;
+ MapVector<Value *, PtrToVectorIndex> PtrVectorIdx;
MapVector<MemTransferInst *, MemTransferInfo> TransferInfo;
} Vector;
struct {
@@ -305,7 +314,6 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
if (!binaryOpIsDerivedFromSameAlloca(AA.Alloca, Cur, SI, 1, 2))
return RejectUser(Inst, "select from mixed objects");
WorkList.push_back(Inst);
- AA.HaveSelectOrPHI = true;
} else if (auto *Phi = dyn_cast<PHINode>(Inst)) {
// Repeat for phis.
@@ -324,7 +332,6 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
}
WorkList.push_back(Inst);
- AA.HaveSelectOrPHI = true;
}
}
}
@@ -467,6 +474,11 @@ static bool isSupportedMemset(MemSetInst *I, AllocaInst *AI,
match(I->getOperand(2), m_SpecificInt(Size)) && !I->isVolatile();
}
+/// Materializes the vector element index that \p Ptr refers to.
+///
+/// Recurses through phis and selects, which contribute an index that is itself
+/// a phi or select. All entries are expected to have been created during
+/// analysis, so this only ever updates them and never invalidates the map.
static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
IRBuilder<> B(Ptr->getContext());
@@ -474,43 +486,88 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
if (Ptr == AA.Alloca)
return B.getInt32(0);
+ auto I = AA.Vector.PtrVectorIdx.find(Ptr);
+ assert(I != AA.Vector.PtrVectorIdx.end() && "Must have entry for pointer!");
+
+ if (Value *Full = I->second.Full)
+ return Full;
+
+ // Phis and selects have no index of their own: theirs is built out of the
+ // indices of their operands.
+ if (auto *Phi = dyn_cast<PHINode>(Ptr)) {
+ // Memoize the new phi before recursing. The pointer phi may be
+ // loop-carried, in which case resolving an incoming value comes back here.
+ PHINode *NewPhi =
+ PHINode::Create(B.getInt32Ty(), Phi->getNumIncomingValues(),
+ Phi->getName() + ".vecidx", Phi->getIterator());
+ I->second.Full = NewPhi;
+
+ for (unsigned Idx = 0, E = Phi->getNumIncomingValues(); Idx != E; ++Idx) {
+ // SSA guarantees the incoming value dominates the end of the incoming
+ // block, and each index is materialized at its own pointer's definition,
+ // so the index dominates the edge as well.
+ Value *IncomingIdx = calculateVectorIndex(Phi->getIncomingValue(Idx), AA);
+ NewPhi->addIncoming(IncomingIdx, Phi->getIncomingBlock(Idx));
+ }
+ return NewPhi;
+ }
+
+ if (auto *Sel = dyn_cast<SelectInst>(Ptr)) {
+ Value *TrueIdx = calculateVectorIndex(Sel->getTrueValue(), AA);
+ Value *FalseIdx = calculateVectorIndex(Sel->getFalseValue(), AA);
+ B.SetInsertPoint(Sel);
+ Value *Result = B.CreateSelect(Sel->getCondition(), TrueIdx, FalseIdx,
+ Sel->getName() + ".vecidx");
+ // Re-find rather than reusing I, which was obtained before recursing.
+ AA.Vector.PtrVectorIdx.find(Ptr)->second.Full = Result;
+ return Result;
+ }
+
auto *GEP = cast<GetElementPtrInst>(Ptr);
- auto I = AA.Vector.GEPVectorIdx.find(GEP);
- assert(I != AA.Vector.GEPVectorIdx.end() && "Must have entry for GEP!");
- if (!I->second.Full) {
- Value *Result = nullptr;
- B.SetInsertPoint(GEP);
+ // Resolve the base index first, as this may create instructions and
+ // invalidate iterators into the map.
+ Value *BaseIdx =
+ I->second.Base ? calculateVectorIndex(I->second.Base, AA) : nullptr;
- if (I->second.VarIndex) {
- Result = I->second.VarIndex;
- Result = B.CreateSExtOrTrunc(Result, B.getInt32Ty());
+ PtrToVectorIndex &Entry = AA.Vector.PtrVectorIdx.find(Ptr)->second;
+ Value *Result = nullptr;
+ B.SetInsertPoint(GEP);
- if (I->second.VarMul)
- Result = B.CreateMul(Result, I->second.VarMul);
+ if (Entry.VarIndex) {
+ Result = Entry.VarIndex;
+ Result = B.CreateSExtOrTrunc(Result, B.getInt32Ty());
- if (I->second.VarShift)
- Result = B.CreateAShr(Result, I->second.VarShift, "", /*isExact*/ true);
- }
+ if (Entry.VarMul)
+ Result = B.CreateMul(Result, Entry.VarMul);
- if (I->second.ConstIndex) {
- if (Result)
- Result = B.CreateAdd(Result, I->second.ConstIndex);
- else
- Result = I->second.ConstIndex;
- }
+ if (Entry.VarShift)
+ Result = B.CreateAShr(Result, Entry.VarShift, "", /*isExact*/ true);
+ }
- if (!Result)
- Result = B.getInt32(0);
+ if (Entry.ConstIndex) {
+ if (Result)
+ Result = B.CreateAdd(Result, Entry.ConstIndex);
+ else
+ Result = Entry.ConstIndex;
+ }
- I->second.Full = Result;
+ if (BaseIdx) {
+ // A pointer with a Base has no variable offset of its own, so this is at
+ // most a constant added to the base index.
+ assert(!Entry.VarIndex && "Base already occupies the variable slot");
+ Result = Result ? B.CreateAdd(BaseIdx, Result) : BaseIdx;
}
- return I->second.Full;
+ if (!Result)
+ Result = B.getInt32(0);
+
+ Entry.Full = Result;
+ return Result;
}
-static std::optional<GEPToVectorIndex>
-computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
+static std::optional<PtrToVectorIndex>
+computeGEPToVectorIndex(GetElementPtrInst *GEP, const AllocaAnalysis &AA,
Type *VecElemTy, const DataLayout &DL) {
// TODO: Extracting a "multiple of X" from a GEP might be a useful generic
// helper.
@@ -545,7 +602,18 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
CurPtr = CurGEP->getPointerOperand();
}
- assert(CurPtr == Alloca && "GEP not based on alloca");
+ // The walk ends either at the alloca itself or at a phi/select that merges
+ // pointers into the alloca, whose index is resolved separately.
+ Value *Base = nullptr;
+ if (CurPtr != AA.Alloca) {
+ if (!isa<PHINode, SelectInst>(CurPtr) || !AA.Pointers.contains(CurPtr))
+ return {};
+ // The base index occupies the single variable slot, so this GEP can only
+ // add a constant of its own.
+ if (!VarOffsets.empty())
+ return {};
+ Base = CurPtr;
+ }
int64_t VecElemSize = DL.getTypeAllocSize(VecElemTy);
if (VarOffsets.size() > 1)
@@ -558,7 +626,8 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
return {};
APInt IndexQuot = ConstOffset.sdiv(VecElemSize);
- GEPToVectorIndex Result;
+ PtrToVectorIndex Result;
+ Result.Base = Base;
if (!ConstOffset.isZero())
Result.ConstIndex = ConstantInt::get(Ctx, IndexQuot.sextOrTrunc(BW));
@@ -981,11 +1050,6 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
}
void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
- if (AA.HaveSelectOrPHI) {
- LLVM_DEBUG(dbgs() << " Cannot convert to vector due to select or phi\n");
- return;
- }
-
Type *AllocaTy = AA.Alloca->getAllocatedType();
AA.Vector.Ty = getVectorTypeForAlloca(AllocaTy);
if (!AA.Vector.Ty)
@@ -1038,12 +1102,32 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
if (auto *GEP = dyn_cast<GetElementPtrInst>(Inst)) {
// If we can't compute a vector index from this GEP, then we can't
// promote this alloca to vector.
- auto Index = computeGEPToVectorIndex(GEP, AA.Alloca, VecEltTy, DL);
+ auto Index = computeGEPToVectorIndex(GEP, AA, VecEltTy, DL);
if (!Index)
return RejectUser(Inst, "cannot compute vector index for GEP");
- AA.Vector.GEPVectorIdx[GEP] = std::move(Index.value());
- AA.Vector.UsersToRemove.push_back(Inst);
+ AA.Vector.PtrVectorIdx[GEP] = std::move(Index.value());
+ AA.Vector.UsersToRemove.insert(Inst);
+ continue;
+ }
+
+ // A phi or select merging pointers into this alloca becomes a phi or select
+ // of their vector indices. collectAllocaUses has already checked that every
+ // operand is derived from this alloca, but that also admits null, for which
+ // there is no index.
+ if (isa<PHINode, SelectInst>(Inst)) {
+ for (Value *Op : Inst->operands()) {
+ if (!Op->getType()->isPointerTy())
+ continue;
+ Value *Ptr = Op->stripPointerCasts();
+ if (Ptr != AA.Alloca && !AA.Pointers.contains(Ptr))
+ return RejectUser(Inst, "operand is not derived from this alloca");
+ }
+
+ // The index itself is built during promotion; this only reserves the
+ // entry so the commit phase never has to insert into the map.
+ AA.Vector.PtrVectorIdx.insert({Inst, PtrToVectorIndex{}});
+ AA.Vector.UsersToRemove.insert(Inst);
continue;
}
@@ -1066,9 +1150,11 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
if (Ptr == AA.Alloca)
return ConstantInt::get(Ptr->getContext(), APInt(32, 0));
- GetElementPtrInst *GEP = cast<GetElementPtrInst>(Ptr);
- const auto &GEPI = AA.Vector.GEPVectorIdx.find(GEP)->second;
- if (GEPI.VarIndex)
+ auto *GEP = dyn_cast<GetElementPtrInst>(Ptr);
+ if (!GEP)
+ return nullptr;
+ const auto &GEPI = AA.Vector.PtrVectorIdx.find(GEP)->second;
+ if (GEPI.VarIndex || GEPI.Base)
return nullptr;
if (GEPI.ConstIndex)
return GEPI.ConstIndex;
@@ -1106,14 +1192,14 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
if (isAssumeLikeIntrinsic(Inst)) {
if (!Inst->use_empty())
return RejectUser(Inst, "assume-like intrinsic cannot have any users");
- AA.Vector.UsersToRemove.push_back(Inst);
+ AA.Vector.UsersToRemove.insert(Inst);
continue;
}
if (isa<ICmpInst>(Inst) && all_of(Inst->users(), [](User *U) {
return isAssumeLikeIntrinsic(cast<Instruction>(U));
})) {
- AA.Vector.UsersToRemove.push_back(Inst);
+ AA.Vector.UsersToRemove.insert(Inst);
continue;
}
@@ -1202,12 +1288,22 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
I->eraseFromParent();
}
- // Delete all the users that are known to be removeable.
- for (Instruction *I : reverse(AA.Vector.UsersToRemove)) {
+ // Delete all the users that are known to be removeable. A loop-carried
+ // pointer phi forms a use cycle with the pointers derived from it, so break
+ // all the references first rather than relying on a deletion order.
+ for (Instruction *I : AA.Vector.UsersToRemove) {
I->dropDroppableUses();
- assert(I->use_empty());
- I->eraseFromParent();
+ assert(all_of(I->users(),
+ [&](User *U) {
+ return AA.Vector.UsersToRemove.contains(
+ cast<Instruction>(U));
+ }) &&
+ "Removed instruction still used outside the removed set");
}
+ for (Instruction *I : AA.Vector.UsersToRemove)
+ I->replaceAllUsesWith(PoisonValue::get(I->getType()));
+ for (Instruction *I : AA.Vector.UsersToRemove)
+ I->eraseFromParent();
// Alloca should now be dead too.
assert(AA.Alloca->use_empty());
@@ -1386,11 +1482,11 @@ bool AMDGPUPromoteAllocaImpl::binaryOpIsDerivedFromSameAlloca(
bool AMDGPUPromoteAllocaImpl::allOpsAreDerivedFromSameAlloca(
Value *Alloca, Value *Val, Instruction *Inst) const {
- for (int i = 0; i < Inst->getNumOperands(); i++) {
- Value *Op = Inst->getOperand(i);
+ for (unsigned I = 0, E = Inst->getNumOperands(); I != E; ++I) {
+ Value *Op = Inst->getOperand(I);
if (Op == Val)
continue;
- if (!binaryOpIsDerivedFromSameAlloca(Alloca, Op, Inst, i, i))
+ if (!binaryOpIsDerivedFromSameAlloca(Alloca, Op, Inst, I, I))
return false;
}
return true;
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
index b38c2d5b205653..df8d8376cc8dc2 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
@@ -1,5 +1,8 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca < %s | FileCheck %s
+; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-vector < %s | FileCheck %s
+
+; Selects are promotable to vector too, so pin this test to the LDS path.
+; Vector promotion of selects is covered by promote-alloca-vector-select.ll.
define amdgpu_kernel void @lds_promoted_alloca_select_invalid_pointer_operand() #0 {
; CHECK-LABEL: define amdgpu_kernel void @lds_promoted_alloca_select_invalid_pointer_operand(
>From 0513170e5ab9f7bd80dd12ec97f8648989370ceb Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Thu, 24 Sep 2026 23:09:27 +0000
Subject: [PATCH 3/6] Clarifications
---
llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 18 ++++++++++++------
1 file changed, 12 insertions(+), 6 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index 711f90e0762c49..b9152458e78503 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -328,7 +328,7 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
break;
default:
if (!allOpsAreDerivedFromSameAlloca(AA.Alloca, Cur, Phi))
- return RejectUser(Inst, "phi with too many operands");
+ return RejectUser(Inst, "phi from mixed objects");
}
WorkList.push_back(Inst);
@@ -522,7 +522,7 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
AA.Vector.PtrVectorIdx.find(Ptr)->second.Full = Result;
return Result;
}
-
+ // If it was not a phi or select, it must be a GEP.
auto *GEP = cast<GetElementPtrInst>(Ptr);
// Resolve the base index first, as this may create instructions and
@@ -530,6 +530,7 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
Value *BaseIdx =
I->second.Base ? calculateVectorIndex(I->second.Base, AA) : nullptr;
+ // Re-find by Ptr to get the most up to date entry.
PtrToVectorIndex &Entry = AA.Vector.PtrVectorIdx.find(Ptr)->second;
Value *Result = nullptr;
B.SetInsertPoint(GEP);
@@ -1116,12 +1117,17 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
// operand is derived from this alloca, but that also admits null, for which
// there is no index.
if (isa<PHINode, SelectInst>(Inst)) {
- for (Value *Op : Inst->operands()) {
- if (!Op->getType()->isPointerTy())
- continue;
+ // Check the merged pointers only. A phi's operands are all incoming
+ // values, but a select's first operand is its condition.
+ auto PtrOps = isa<PHINode>(Inst)
+ ? cast<PHINode>(Inst)->incoming_values()
+ : make_range(Inst->op_begin() + 1, Inst->op_end());
+ for (Value *Op : PtrOps) {
+ // collectAllocaUses has already checked that these derive from this
+ // alloca, but it also admits null, which has no vector index.
Value *Ptr = Op->stripPointerCasts();
if (Ptr != AA.Alloca && !AA.Pointers.contains(Ptr))
- return RejectUser(Inst, "operand is not derived from this alloca");
+ return RejectUser(Inst, "operand has no vector index");
}
// The index itself is built during promotion; this only reserves the
>From 8cbb286b761008484059fddaf4bf32bb14336202 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Fri, 25 Sep 2026 03:51:59 +0000
Subject: [PATCH 4/6] Adding lit-tests
---
.../promote-alloca-select-vector-preferred.ll | 131 ++++++++
.../AMDGPU/promote-alloca-vector-phi.ll | 298 ++++++++++++++++++
.../AMDGPU/promote-alloca-vector-select.ll | 96 ++++++
3 files changed, 525 insertions(+)
create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll
create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-vector-select.ll
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll
new file mode 100644
index 00000000000000..5def4376976c18
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll
@@ -0,0 +1,131 @@
+; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca < %s | FileCheck %s
+
+; Same functions as promote-alloca-to-lds-select.ll, but with the default
+; pipeline, where promotion to vector is tried before promotion to LDS. Only the
+; allocas that vector promotion turns down fall through to LDS.
+
+; CHECK-LABEL: @lds_promoted_alloca_select_invalid_pointer_operand(
+; CHECK: %alloca = alloca i32
+; CHECK: select i1 undef, ptr addrspace(5) poison, ptr addrspace(5) %alloca
+define amdgpu_kernel void @lds_promoted_alloca_select_invalid_pointer_operand() #0 {
+ %alloca = alloca i32, align 4, addrspace(5)
+ %select = select i1 undef, ptr addrspace(5) poison, ptr addrspace(5) %alloca
+ store i32 0, ptr addrspace(5) %select, align 4
+ ret void
+}
+
+; The select of two pointers becomes a select of the two vector indices.
+; CHECK-LABEL: @lds_promote_alloca_select_two_derived_pointers(
+; CHECK: %alloca = freeze <16 x i32> poison
+; CHECK: %select.vecidx = select i1 undef, i32 %a, i32 %b
+; CHECK: insertelement <16 x i32> %alloca, i32 0, i32 %select.vecidx
+define amdgpu_kernel void @lds_promote_alloca_select_two_derived_pointers(i32 %a, i32 %b) #0 {
+ %alloca = alloca [16 x i32], align 4, addrspace(5)
+ %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
+ %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
+ %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
+ store i32 0, ptr addrspace(5) %select, align 4
+ ret void
+}
+
+; FIXME: This should be promotable but requires knowing that both will be promoted first.
+
+; CHECK-LABEL: @lds_promote_alloca_select_two_allocas(
+; CHECK: %alloca0 = alloca i32, i32 16, align 4
+; CHECK: %alloca1 = alloca i32, i32 16, align 4
+; CHECK: %ptr0 = getelementptr inbounds i32, ptr addrspace(5) %alloca0, i32 %a
+; CHECK: %ptr1 = getelementptr inbounds i32, ptr addrspace(5) %alloca1, i32 %b
+; CHECK: %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
+define amdgpu_kernel void @lds_promote_alloca_select_two_allocas(i32 %a, i32 %b) #0 {
+ %alloca0 = alloca i32, i32 16, align 4, addrspace(5)
+ %alloca1 = alloca i32, i32 16, align 4, addrspace(5)
+ %ptr0 = getelementptr inbounds i32, ptr addrspace(5) %alloca0, i32 %a
+ %ptr1 = getelementptr inbounds i32, ptr addrspace(5) %alloca1, i32 %b
+ %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
+ store i32 0, ptr addrspace(5) %select, align 4
+ ret void
+}
+
+; Both indices are constant, so the select of indices folds away entirely.
+; CHECK-LABEL: @lds_promote_alloca_select_two_derived_constant_pointers(
+; CHECK: %alloca = freeze <16 x i32> poison
+; CHECK-NOT: select
+; CHECK: insertelement <16 x i32> %alloca, i32 0, i32 3
+define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointers() #0 {
+ %alloca = alloca [16 x i32], align 4, addrspace(5)
+ %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 1
+ %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 3
+ %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
+ store i32 0, ptr addrspace(5) %select, align 4
+ ret void
+}
+
+; FIXME: Can be promoted, but we'd have to recursively show that the select
+; operands all point to the same alloca.
+
+; CHECK-LABEL: @lds_promoted_alloca_select_input_select(
+; CHECK: alloca
+define amdgpu_kernel void @lds_promoted_alloca_select_input_select(i32 %a, i32 %b, i32 %c, i1 %c1, i1 %c2) #0 {
+ %alloca = alloca [16 x i32], align 4, addrspace(5)
+ %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
+ %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
+ %ptr2 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %c
+ %select0 = select i1 %c1, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
+ %select1 = select i1 %c2, ptr addrspace(5) %select0, ptr addrspace(5) %ptr2
+ store i32 0, ptr addrspace(5) %select1, align 4
+ ret void
+}
+
+define amdgpu_kernel void @lds_promoted_alloca_select_input_phi(i32 %a, i32 %b, i32 %c, i1 %c0) #0 {
+entry:
+ %alloca = alloca [16 x i32], align 4, addrspace(5)
+ %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
+ %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
+ store i32 0, ptr addrspace(5) %ptr0
+ br i1 %c0, label %bb1, label %bb2
+
+bb1:
+ %ptr2 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %c
+ %select0 = select i1 undef, ptr addrspace(5) poison, ptr addrspace(5) %ptr2
+ store i32 0, ptr addrspace(5) %ptr1
+ br label %bb2
+
+bb2:
+ %phi.ptr = phi ptr addrspace(5) [ %ptr0, %entry ], [ %select0, %bb1 ]
+ %select1 = select i1 undef, ptr addrspace(5) %phi.ptr, ptr addrspace(5) %ptr1
+ store i32 0, ptr addrspace(5) %select1, align 4
+ ret void
+}
+
+; CHECK-LABEL: @select_null_rhs(
+; CHECK-NOT: alloca
+; CHECK: select i1 %tmp2, ptr addrspace(3) %{{[0-9]+}}, ptr addrspace(3) null
+define amdgpu_kernel void @select_null_rhs(ptr addrspace(1) nocapture %arg, i32 %arg1) #1 {
+bb:
+ %tmp = alloca double, align 8, addrspace(5)
+ store double 0.000000e+00, ptr addrspace(5) %tmp, align 8
+ %tmp2 = icmp eq i32 %arg1, 0
+ %tmp3 = select i1 %tmp2, ptr addrspace(5) %tmp, ptr addrspace(5) zeroinitializer
+ store double 1.000000e+00, ptr addrspace(5) %tmp3, align 8
+ %tmp4 = load double, ptr addrspace(5) %tmp, align 8
+ store double %tmp4, ptr addrspace(1) %arg
+ ret void
+}
+
+; CHECK-LABEL: @select_null_lhs(
+; CHECK-NOT: alloca
+; CHECK: select i1 %tmp2, ptr addrspace(3) null, ptr addrspace(3) %{{[0-9]+}}
+define amdgpu_kernel void @select_null_lhs(ptr addrspace(1) nocapture %arg, i32 %arg1) #1 {
+bb:
+ %tmp = alloca double, align 8, addrspace(5)
+ store double 0.000000e+00, ptr addrspace(5) %tmp, align 8
+ %tmp2 = icmp eq i32 %arg1, 0
+ %tmp3 = select i1 %tmp2, ptr addrspace(5) zeroinitializer, ptr addrspace(5) %tmp
+ store double 1.000000e+00, ptr addrspace(5) %tmp3, align 8
+ %tmp4 = load double, ptr addrspace(5) %tmp, align 8
+ store double %tmp4, ptr addrspace(1) %arg
+ ret void
+}
+
+attributes #0 = { norecurse nounwind "amdgpu-waves-per-eu"="1,1" "amdgpu-flat-work-group-size"="1,256" }
+attributes #1 = { norecurse nounwind }
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
new file mode 100644
index 00000000000000..fcfb6253408706
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
@@ -0,0 +1,298 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu-amd-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-lds < %s | FileCheck %s
+
+; A phi merging pointers derived from a single alloca becomes a phi of the
+; corresponding vector indices. LDS promotion is disabled so that allocas which
+; cannot be vectorized are left alone.
+
+; Both incoming values are constant offsets into the alloca, so the index phi is
+; a phi of constants.
+define amdgpu_kernel void @phi_const_geps(i1 %cond, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_const_geps(
+; CHECK-SAME: i1 [[COND:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT: br i1 [[COND]], label %[[IF:.*]], label %[[ELSE:.*]]
+; CHECK: [[IF]]:
+; CHECK-NEXT: br label %[[ENDIF:.*]]
+; CHECK: [[ELSE]]:
+; CHECK-NEXT: br label %[[ENDIF]]
+; CHECK: [[ENDIF]]:
+; CHECK-NEXT: [[PTR_VECIDX:%.*]] = phi i32 [ 1, %[[IF]] ], [ 2, %[[ELSE]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x i32> [[TMP0]], i32 [[PTR_VECIDX]]
+; CHECK-NEXT: store i32 [[TMP1]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [4 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ %gep1 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+ %gep2 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+ br i1 %cond, label %if, label %else
+
+if:
+ br label %endif
+
+else:
+ br label %endif
+
+endif:
+ %ptr = phi ptr addrspace(5) [ %gep1, %if ], [ %gep2, %else ]
+ %val = load i32, ptr addrspace(5) %ptr, align 4
+ store i32 %val, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; Three incoming values, one of which is the alloca itself (index 0), with a
+; further constant GEP applied to the phi result. This is the shape SROA's
+; gep(phi) -> phi(gep) fold leaves behind.
+define amdgpu_kernel void @phi_gep_and_alloca_then_gep(i32 %sel, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_gep_and_alloca_then_gep(
+; CHECK-SAME: i32 [[SEL:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT: switch i32 [[SEL]], label %[[C:.*]] [
+; CHECK-NEXT: i32 0, label %[[A:.*]]
+; CHECK-NEXT: i32 1, label %[[B:.*]]
+; CHECK-NEXT: ]
+; CHECK: [[A]]:
+; CHECK-NEXT: br label %[[JOIN:.*]]
+; CHECK: [[B]]:
+; CHECK-NEXT: br label %[[JOIN]]
+; CHECK: [[C]]:
+; CHECK-NEXT: br label %[[JOIN]]
+; CHECK: [[JOIN]]:
+; CHECK-NEXT: [[PTR_VECIDX:%.*]] = phi i32 [ 2, %[[A]] ], [ 2, %[[B]] ], [ 0, %[[C]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = add i32 [[PTR_VECIDX]], 1
+; CHECK-NEXT: [[TMP2:%.*]] = extractelement <4 x i32> [[TMP0]], i32 [[TMP1]]
+; CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [2 x [2 x i32]], align 16, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ %gep8 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+ switch i32 %sel, label %c [
+ i32 0, label %a
+ i32 1, label %b
+ ]
+
+a:
+ br label %join
+
+b:
+ br label %join
+
+c:
+ br label %join
+
+join:
+ %ptr = phi ptr addrspace(5) [ %gep8, %a ], [ %gep8, %b ], [ %alloca, %c ]
+ %gep4 = getelementptr inbounds i8, ptr addrspace(5) %ptr, i32 4
+ %val = load i32, ptr addrspace(5) %gep4, align 4
+ store i32 %val, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; Variable indices flow through into the index phi.
+define amdgpu_kernel void @phi_var_geps(i1 %cond, i32 %a, i32 %b, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_var_geps(
+; CHECK-SAME: i1 [[COND:%.*]], i32 [[A:%.*]], i32 [[B:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT: br i1 [[COND]], label %[[IF:.*]], label %[[ELSE:.*]]
+; CHECK: [[IF]]:
+; CHECK-NEXT: br label %[[ENDIF:.*]]
+; CHECK: [[ELSE]]:
+; CHECK-NEXT: br label %[[ENDIF]]
+; CHECK: [[ENDIF]]:
+; CHECK-NEXT: [[PTR_VECIDX:%.*]] = phi i32 [ [[A]], %[[IF]] ], [ [[B]], %[[ELSE]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x i32> [[TMP0]], i32 [[PTR_VECIDX]]
+; CHECK-NEXT: store i32 [[TMP1]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [4 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ %gepa = getelementptr inbounds [4 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
+ %gepb = getelementptr inbounds [4 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
+ br i1 %cond, label %if, label %else
+
+if:
+ br label %endif
+
+else:
+ br label %endif
+
+endif:
+ %ptr = phi ptr addrspace(5) [ %gepa, %if ], [ %gepb, %else ]
+ %val = load i32, ptr addrspace(5) %ptr, align 4
+ store i32 %val, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; A store through the phi has to write back into the promoted vector.
+define amdgpu_kernel void @phi_store_through_phi(i1 %cond, i32 %val, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_store_through_phi(
+; CHECK-SAME: i1 [[COND:%.*]], i32 [[VAL:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT: br i1 [[COND]], label %[[IF:.*]], label %[[ELSE:.*]]
+; CHECK: [[IF]]:
+; CHECK-NEXT: br label %[[ENDIF:.*]]
+; CHECK: [[ELSE]]:
+; CHECK-NEXT: br label %[[ENDIF]]
+; CHECK: [[ENDIF]]:
+; CHECK-NEXT: [[PTR_VECIDX:%.*]] = phi i32 [ 1, %[[IF]] ], [ 2, %[[ELSE]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <4 x i32> [[TMP0]], i32 [[VAL]], i32 [[PTR_VECIDX]]
+; CHECK-NEXT: [[TMP2:%.*]] = extractelement <4 x i32> [[TMP1]], i32 0
+; CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [4 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ %gep1 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+ %gep2 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+ br i1 %cond, label %if, label %else
+
+if:
+ br label %endif
+
+else:
+ br label %endif
+
+endif:
+ %ptr = phi ptr addrspace(5) [ %gep1, %if ], [ %gep2, %else ]
+ store i32 %val, ptr addrspace(5) %ptr, align 4
+ %reload = load i32, ptr addrspace(5) %alloca, align 4
+ store i32 %reload, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; Negative: the phi mixes two different allocas, so neither is promotable.
+define amdgpu_kernel void @phi_two_allocas(i1 %cond, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_two_allocas(
+; CHECK-SAME: i1 [[COND:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA0:%.*]] = alloca [4 x i32], align 4, addrspace(5)
+; CHECK-NEXT: [[ALLOCA1:%.*]] = alloca [4 x i32], align 4, addrspace(5)
+; CHECK-NEXT: store i32 7, ptr addrspace(5) [[ALLOCA0]], align 4
+; CHECK-NEXT: store i32 9, ptr addrspace(5) [[ALLOCA1]], align 4
+; CHECK-NEXT: br i1 [[COND]], label %[[IF:.*]], label %[[ELSE:.*]]
+; CHECK: [[IF]]:
+; CHECK-NEXT: br label %[[ENDIF:.*]]
+; CHECK: [[ELSE]]:
+; CHECK-NEXT: br label %[[ENDIF]]
+; CHECK: [[ENDIF]]:
+; CHECK-NEXT: [[PTR:%.*]] = phi ptr addrspace(5) [ [[ALLOCA0]], %[[IF]] ], [ [[ALLOCA1]], %[[ELSE]] ]
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr addrspace(5) [[PTR]], align 4
+; CHECK-NEXT: store i32 [[VAL]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca0 = alloca [4 x i32], align 4, addrspace(5)
+ %alloca1 = alloca [4 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca0, align 4
+ store i32 9, ptr addrspace(5) %alloca1, align 4
+ br i1 %cond, label %if, label %else
+
+if:
+ br label %endif
+
+else:
+ br label %endif
+
+endif:
+ %ptr = phi ptr addrspace(5) [ %alloca0, %if ], [ %alloca1, %else ]
+ %val = load i32, ptr addrspace(5) %ptr, align 4
+ store i32 %val, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; Negative: a variable-offset GEP applied to the phi would need two variable
+; terms in the index expression, so this must bail rather than assert.
+define amdgpu_kernel void @phi_var_gep_off_phi(i1 %cond, i32 %idx, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_var_gep_off_phi(
+; CHECK-SAME: i1 [[COND:%.*]], i32 [[IDX:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA:%.*]] = alloca [4 x i32], align 4, addrspace(5)
+; CHECK-NEXT: store i32 7, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT: [[GEP1:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 4
+; CHECK-NEXT: [[GEP2:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 8
+; CHECK-NEXT: br i1 [[COND]], label %[[IF:.*]], label %[[ELSE:.*]]
+; CHECK: [[IF]]:
+; CHECK-NEXT: br label %[[ENDIF:.*]]
+; CHECK: [[ELSE]]:
+; CHECK-NEXT: br label %[[ENDIF]]
+; CHECK: [[ENDIF]]:
+; CHECK-NEXT: [[PTR:%.*]] = phi ptr addrspace(5) [ [[GEP1]], %[[IF]] ], [ [[GEP2]], %[[ELSE]] ]
+; CHECK-NEXT: [[VGEP:%.*]] = getelementptr inbounds i32, ptr addrspace(5) [[PTR]], i32 [[IDX]]
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr addrspace(5) [[VGEP]], align 4
+; CHECK-NEXT: store i32 [[VAL]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [4 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ %gep1 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+ %gep2 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+ br i1 %cond, label %if, label %else
+
+if:
+ br label %endif
+
+else:
+ br label %endif
+
+endif:
+ %ptr = phi ptr addrspace(5) [ %gep1, %if ], [ %gep2, %else ]
+ %vgep = getelementptr inbounds i32, ptr addrspace(5) %ptr, i32 %idx
+ %val = load i32, ptr addrspace(5) %vgep, align 4
+ store i32 %val, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; Negative: a loop-carried pointer phi is still rejected up front, because
+; getUnderlyingObject cannot see through the phi to the alloca.
+define amdgpu_kernel void @phi_loop_carried(i32 %n, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_loop_carried(
+; CHECK-SAME: i32 [[N:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[ALLOCA:%.*]] = alloca [4 x i32], align 4, addrspace(5)
+; CHECK-NEXT: store i32 7, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[I:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[PTR:%.*]] = phi ptr addrspace(5) [ [[ALLOCA]], %[[ENTRY]] ], [ [[PTR_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr addrspace(5) [[PTR]], align 4
+; CHECK-NEXT: store i32 [[VAL]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: [[PTR_NEXT]] = getelementptr inbounds i8, ptr addrspace(5) [[PTR]], i32 4
+; CHECK-NEXT: [[I_NEXT]] = add i32 [[I]], 1
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i32 [[I_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [4 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ br label %loop
+
+loop:
+ %i = phi i32 [ 0, %entry ], [ %i.next, %loop ]
+ %ptr = phi ptr addrspace(5) [ %alloca, %entry ], [ %ptr.next, %loop ]
+ %val = load i32, ptr addrspace(5) %ptr, align 4
+ store i32 %val, ptr addrspace(1) %out, align 4
+ %ptr.next = getelementptr inbounds i8, ptr addrspace(5) %ptr, i32 4
+ %i.next = add i32 %i, 1
+ %done = icmp eq i32 %i.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-select.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-select.ll
new file mode 100644
index 00000000000000..e9dac6bff51930
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-select.ll
@@ -0,0 +1,96 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu-amd-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-lds < %s | FileCheck %s
+
+; A select between pointers derived from a single alloca becomes a select of the
+; corresponding vector indices. LDS promotion is disabled so that allocas which
+; cannot be vectorized are left alone.
+
+define amdgpu_kernel void @select_const_geps(i1 %cond, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @select_const_geps(
+; CHECK-SAME: i1 [[COND:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT: [[PTR_VECIDX:%.*]] = select i1 [[COND]], i32 1, i32 2
+; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x i32> [[TMP0]], i32 [[PTR_VECIDX]]
+; CHECK-NEXT: store i32 [[TMP1]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [4 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ %gep1 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+ %gep2 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+ %ptr = select i1 %cond, ptr addrspace(5) %gep1, ptr addrspace(5) %gep2
+ %val = load i32, ptr addrspace(5) %ptr, align 4
+ store i32 %val, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+define amdgpu_kernel void @select_var_geps(i1 %cond, i32 %a, i32 %b, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @select_var_geps(
+; CHECK-SAME: i1 [[COND:%.*]], i32 [[A:%.*]], i32 [[B:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT: [[PTR_VECIDX:%.*]] = select i1 [[COND]], i32 [[A]], i32 [[B]]
+; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x i32> [[TMP0]], i32 [[PTR_VECIDX]]
+; CHECK-NEXT: store i32 [[TMP1]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [4 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ %gepa = getelementptr inbounds [4 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
+ %gepb = getelementptr inbounds [4 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
+ %ptr = select i1 %cond, ptr addrspace(5) %gepa, ptr addrspace(5) %gepb
+ %val = load i32, ptr addrspace(5) %ptr, align 4
+ store i32 %val, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; One side is the alloca itself, and a constant GEP is applied to the result.
+define amdgpu_kernel void @select_alloca_then_gep(i1 %cond, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @select_alloca_then_gep(
+; CHECK-SAME: i1 [[COND:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT: [[PTR_VECIDX:%.*]] = select i1 [[COND]], i32 2, i32 0
+; CHECK-NEXT: [[TMP1:%.*]] = add i32 [[PTR_VECIDX]], 1
+; CHECK-NEXT: [[TMP2:%.*]] = extractelement <4 x i32> [[TMP0]], i32 [[TMP1]]
+; CHECK-NEXT: store i32 [[TMP2]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [4 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ %gep8 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+ %ptr = select i1 %cond, ptr addrspace(5) %gep8, ptr addrspace(5) %alloca
+ %gep4 = getelementptr inbounds i8, ptr addrspace(5) %ptr, i32 4
+ %val = load i32, ptr addrspace(5) %gep4, align 4
+ store i32 %val, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; Negative: null has no vector index, so this is not vector promotable even
+; though it is accepted for LDS promotion.
+define amdgpu_kernel void @select_null_operand(i1 %cond, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @select_null_operand(
+; CHECK-SAME: i1 [[COND:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA:%.*]] = alloca [4 x i32], align 4, addrspace(5)
+; CHECK-NEXT: store i32 7, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT: [[PTR:%.*]] = select i1 [[COND]], ptr addrspace(5) [[ALLOCA]], ptr addrspace(5) null
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr addrspace(5) [[PTR]], align 4
+; CHECK-NEXT: store i32 [[VAL]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [4 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ %ptr = select i1 %cond, ptr addrspace(5) %alloca, ptr addrspace(5) null
+ %val = load i32, ptr addrspace(5) %ptr, align 4
+ store i32 %val, ptr addrspace(1) %out, align 4
+ ret void
+}
>From e0c5528832de201b4508c88b0ecd77cd14acd974 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Fri, 25 Sep 2026 17:35:58 +0000
Subject: [PATCH 5/6] Remove tests promotable to vector from
promote-alloca-to-lds-select.ll
---
.../promote-alloca-select-vector-preferred.ll | 131 ------------------
.../AMDGPU/promote-alloca-to-lds-select.ll | 76 +---------
2 files changed, 5 insertions(+), 202 deletions(-)
delete mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll
deleted file mode 100644
index 5def4376976c18..00000000000000
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll
+++ /dev/null
@@ -1,131 +0,0 @@
-; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca < %s | FileCheck %s
-
-; Same functions as promote-alloca-to-lds-select.ll, but with the default
-; pipeline, where promotion to vector is tried before promotion to LDS. Only the
-; allocas that vector promotion turns down fall through to LDS.
-
-; CHECK-LABEL: @lds_promoted_alloca_select_invalid_pointer_operand(
-; CHECK: %alloca = alloca i32
-; CHECK: select i1 undef, ptr addrspace(5) poison, ptr addrspace(5) %alloca
-define amdgpu_kernel void @lds_promoted_alloca_select_invalid_pointer_operand() #0 {
- %alloca = alloca i32, align 4, addrspace(5)
- %select = select i1 undef, ptr addrspace(5) poison, ptr addrspace(5) %alloca
- store i32 0, ptr addrspace(5) %select, align 4
- ret void
-}
-
-; The select of two pointers becomes a select of the two vector indices.
-; CHECK-LABEL: @lds_promote_alloca_select_two_derived_pointers(
-; CHECK: %alloca = freeze <16 x i32> poison
-; CHECK: %select.vecidx = select i1 undef, i32 %a, i32 %b
-; CHECK: insertelement <16 x i32> %alloca, i32 0, i32 %select.vecidx
-define amdgpu_kernel void @lds_promote_alloca_select_two_derived_pointers(i32 %a, i32 %b) #0 {
- %alloca = alloca [16 x i32], align 4, addrspace(5)
- %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
- %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
- %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
- store i32 0, ptr addrspace(5) %select, align 4
- ret void
-}
-
-; FIXME: This should be promotable but requires knowing that both will be promoted first.
-
-; CHECK-LABEL: @lds_promote_alloca_select_two_allocas(
-; CHECK: %alloca0 = alloca i32, i32 16, align 4
-; CHECK: %alloca1 = alloca i32, i32 16, align 4
-; CHECK: %ptr0 = getelementptr inbounds i32, ptr addrspace(5) %alloca0, i32 %a
-; CHECK: %ptr1 = getelementptr inbounds i32, ptr addrspace(5) %alloca1, i32 %b
-; CHECK: %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
-define amdgpu_kernel void @lds_promote_alloca_select_two_allocas(i32 %a, i32 %b) #0 {
- %alloca0 = alloca i32, i32 16, align 4, addrspace(5)
- %alloca1 = alloca i32, i32 16, align 4, addrspace(5)
- %ptr0 = getelementptr inbounds i32, ptr addrspace(5) %alloca0, i32 %a
- %ptr1 = getelementptr inbounds i32, ptr addrspace(5) %alloca1, i32 %b
- %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
- store i32 0, ptr addrspace(5) %select, align 4
- ret void
-}
-
-; Both indices are constant, so the select of indices folds away entirely.
-; CHECK-LABEL: @lds_promote_alloca_select_two_derived_constant_pointers(
-; CHECK: %alloca = freeze <16 x i32> poison
-; CHECK-NOT: select
-; CHECK: insertelement <16 x i32> %alloca, i32 0, i32 3
-define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointers() #0 {
- %alloca = alloca [16 x i32], align 4, addrspace(5)
- %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 1
- %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 3
- %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
- store i32 0, ptr addrspace(5) %select, align 4
- ret void
-}
-
-; FIXME: Can be promoted, but we'd have to recursively show that the select
-; operands all point to the same alloca.
-
-; CHECK-LABEL: @lds_promoted_alloca_select_input_select(
-; CHECK: alloca
-define amdgpu_kernel void @lds_promoted_alloca_select_input_select(i32 %a, i32 %b, i32 %c, i1 %c1, i1 %c2) #0 {
- %alloca = alloca [16 x i32], align 4, addrspace(5)
- %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
- %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
- %ptr2 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %c
- %select0 = select i1 %c1, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
- %select1 = select i1 %c2, ptr addrspace(5) %select0, ptr addrspace(5) %ptr2
- store i32 0, ptr addrspace(5) %select1, align 4
- ret void
-}
-
-define amdgpu_kernel void @lds_promoted_alloca_select_input_phi(i32 %a, i32 %b, i32 %c, i1 %c0) #0 {
-entry:
- %alloca = alloca [16 x i32], align 4, addrspace(5)
- %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
- %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
- store i32 0, ptr addrspace(5) %ptr0
- br i1 %c0, label %bb1, label %bb2
-
-bb1:
- %ptr2 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %c
- %select0 = select i1 undef, ptr addrspace(5) poison, ptr addrspace(5) %ptr2
- store i32 0, ptr addrspace(5) %ptr1
- br label %bb2
-
-bb2:
- %phi.ptr = phi ptr addrspace(5) [ %ptr0, %entry ], [ %select0, %bb1 ]
- %select1 = select i1 undef, ptr addrspace(5) %phi.ptr, ptr addrspace(5) %ptr1
- store i32 0, ptr addrspace(5) %select1, align 4
- ret void
-}
-
-; CHECK-LABEL: @select_null_rhs(
-; CHECK-NOT: alloca
-; CHECK: select i1 %tmp2, ptr addrspace(3) %{{[0-9]+}}, ptr addrspace(3) null
-define amdgpu_kernel void @select_null_rhs(ptr addrspace(1) nocapture %arg, i32 %arg1) #1 {
-bb:
- %tmp = alloca double, align 8, addrspace(5)
- store double 0.000000e+00, ptr addrspace(5) %tmp, align 8
- %tmp2 = icmp eq i32 %arg1, 0
- %tmp3 = select i1 %tmp2, ptr addrspace(5) %tmp, ptr addrspace(5) zeroinitializer
- store double 1.000000e+00, ptr addrspace(5) %tmp3, align 8
- %tmp4 = load double, ptr addrspace(5) %tmp, align 8
- store double %tmp4, ptr addrspace(1) %arg
- ret void
-}
-
-; CHECK-LABEL: @select_null_lhs(
-; CHECK-NOT: alloca
-; CHECK: select i1 %tmp2, ptr addrspace(3) null, ptr addrspace(3) %{{[0-9]+}}
-define amdgpu_kernel void @select_null_lhs(ptr addrspace(1) nocapture %arg, i32 %arg1) #1 {
-bb:
- %tmp = alloca double, align 8, addrspace(5)
- store double 0.000000e+00, ptr addrspace(5) %tmp, align 8
- %tmp2 = icmp eq i32 %arg1, 0
- %tmp3 = select i1 %tmp2, ptr addrspace(5) zeroinitializer, ptr addrspace(5) %tmp
- store double 1.000000e+00, ptr addrspace(5) %tmp3, align 8
- %tmp4 = load double, ptr addrspace(5) %tmp, align 8
- store double %tmp4, ptr addrspace(1) %arg
- ret void
-}
-
-attributes #0 = { norecurse nounwind "amdgpu-waves-per-eu"="1,1" "amdgpu-flat-work-group-size"="1,256" }
-attributes #1 = { norecurse nounwind }
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
index df8d8376cc8dc2..822781c9e38f9c 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-vector < %s | FileCheck %s
+; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca < %s | FileCheck %s
; Selects are promotable to vector too, so pin this test to the LDS path.
; Vector promotion of selects is covered by promote-alloca-vector-select.ll.
@@ -18,38 +18,6 @@ define amdgpu_kernel void @lds_promoted_alloca_select_invalid_pointer_operand()
ret void
}
-define amdgpu_kernel void @lds_promote_alloca_select_two_derived_pointers(i32 %a, i32 %b) #0 {
-; CHECK-LABEL: define amdgpu_kernel void @lds_promote_alloca_select_two_derived_pointers(
-; CHECK-SAME: i32 [[A:%.*]], i32 [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[TMP1:%.*]] = call noalias nonnull dereferenceable(64) ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
-; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 1
-; CHECK-NEXT: [[TMP3:%.*]] = load i32, ptr addrspace(4) [[TMP2]], align 4, !invariant.load [[META0:![0-9]+]]
-; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 2
-; CHECK-NEXT: [[TMP5:%.*]] = load i32, ptr addrspace(4) [[TMP4]], align 4, !range [[RNG1:![0-9]+]], !invariant.load [[META0]]
-; CHECK-NEXT: [[TMP6:%.*]] = lshr i32 [[TMP3]], 16
-; CHECK-NEXT: [[TMP7:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.x()
-; CHECK-NEXT: [[TMP8:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.y()
-; CHECK-NEXT: [[TMP9:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.z()
-; CHECK-NEXT: [[TMP10:%.*]] = mul nuw nsw i32 [[TMP6]], [[TMP5]]
-; CHECK-NEXT: [[TMP11:%.*]] = mul i32 [[TMP10]], [[TMP7]]
-; CHECK-NEXT: [[TMP12:%.*]] = mul nuw nsw i32 [[TMP8]], [[TMP5]]
-; CHECK-NEXT: [[TMP13:%.*]] = add i32 [[TMP11]], [[TMP12]]
-; CHECK-NEXT: [[TMP14:%.*]] = add i32 [[TMP13]], [[TMP9]]
-; CHECK-NEXT: [[TMP15:%.*]] = getelementptr inbounds [256 x [16 x i32]], ptr addrspace(3) @lds_promote_alloca_select_two_derived_pointers.alloca, i32 0, i32 [[TMP14]]
-; CHECK-NEXT: [[PTR0:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 [[A]]
-; CHECK-NEXT: [[PTR1:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 [[B]]
-; CHECK-NEXT: [[SELECT:%.*]] = select i1 poison, ptr addrspace(3) [[PTR0]], ptr addrspace(3) [[PTR1]]
-; CHECK-NEXT: store i32 0, ptr addrspace(3) [[SELECT]], align 4
-; CHECK-NEXT: ret void
-;
- %alloca = alloca [16 x i32], align 4, addrspace(5)
- %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
- %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
- %select = select i1 poison, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
- store i32 0, ptr addrspace(5) %select, align 4
- ret void
-}
-
; FIXME: This should be promotable but requires knowing that both will be promoted first.
define amdgpu_kernel void @lds_promote_alloca_select_two_allocas(i32 %a, i32 %b) #0 {
@@ -72,39 +40,6 @@ define amdgpu_kernel void @lds_promote_alloca_select_two_allocas(i32 %a, i32 %b)
ret void
}
-; TODO: Maybe this should be canonicalized to select on the constant and GEP after.
-define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointers() #0 {
-; CHECK-LABEL: define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointers(
-; CHECK-SAME: ) #[[ATTR0]] {
-; CHECK-NEXT: [[TMP1:%.*]] = call noalias nonnull dereferenceable(64) ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
-; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 1
-; CHECK-NEXT: [[TMP3:%.*]] = load i32, ptr addrspace(4) [[TMP2]], align 4, !invariant.load [[META0]]
-; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 2
-; CHECK-NEXT: [[TMP5:%.*]] = load i32, ptr addrspace(4) [[TMP4]], align 4, !range [[RNG1]], !invariant.load [[META0]]
-; CHECK-NEXT: [[TMP6:%.*]] = lshr i32 [[TMP3]], 16
-; CHECK-NEXT: [[TMP7:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.x()
-; CHECK-NEXT: [[TMP8:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.y()
-; CHECK-NEXT: [[TMP9:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.z()
-; CHECK-NEXT: [[TMP10:%.*]] = mul nuw nsw i32 [[TMP6]], [[TMP5]]
-; CHECK-NEXT: [[TMP11:%.*]] = mul i32 [[TMP10]], [[TMP7]]
-; CHECK-NEXT: [[TMP12:%.*]] = mul nuw nsw i32 [[TMP8]], [[TMP5]]
-; CHECK-NEXT: [[TMP13:%.*]] = add i32 [[TMP11]], [[TMP12]]
-; CHECK-NEXT: [[TMP14:%.*]] = add i32 [[TMP13]], [[TMP9]]
-; CHECK-NEXT: [[TMP15:%.*]] = getelementptr inbounds [256 x [16 x i32]], ptr addrspace(3) @lds_promote_alloca_select_two_derived_constant_pointers.alloca, i32 0, i32 [[TMP14]]
-; CHECK-NEXT: [[PTR0:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 1
-; CHECK-NEXT: [[PTR1:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 3
-; CHECK-NEXT: [[SELECT:%.*]] = select i1 poison, ptr addrspace(3) [[PTR0]], ptr addrspace(3) [[PTR1]]
-; CHECK-NEXT: store i32 0, ptr addrspace(3) [[SELECT]], align 4
-; CHECK-NEXT: ret void
-;
- %alloca = alloca [16 x i32], align 4, addrspace(5)
- %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 1
- %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 3
- %select = select i1 poison, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
- store i32 0, ptr addrspace(5) %select, align 4
- ret void
-}
-
; FIXME: Can be promoted, but we'd have to recursively show that the select
; operands all point to the same alloca.
@@ -176,9 +111,9 @@ define amdgpu_kernel void @select_null_rhs(ptr addrspace(1) nocapture %arg, i32
; CHECK-NEXT: [[BB:.*:]]
; CHECK-NEXT: [[TMP0:%.*]] = call noalias nonnull dereferenceable(64) ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 1
-; CHECK-NEXT: [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4, !invariant.load [[META0]]
+; CHECK-NEXT: [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4, !invariant.load [[META0:![0-9]+]]
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 2
-; CHECK-NEXT: [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG2:![0-9]+]], !invariant.load [[META0]]
+; CHECK-NEXT: [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG1:![0-9]+]], !invariant.load [[META0]]
; CHECK-NEXT: [[TMP5:%.*]] = lshr i32 [[TMP2]], 16
; CHECK-NEXT: [[TMP6:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.x()
; CHECK-NEXT: [[TMP7:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.y()
@@ -216,7 +151,7 @@ define amdgpu_kernel void @select_null_lhs(ptr addrspace(1) nocapture %arg, i32
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 1
; CHECK-NEXT: [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4, !invariant.load [[META0]]
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 2
-; CHECK-NEXT: [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG2]], !invariant.load [[META0]]
+; CHECK-NEXT: [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG1]], !invariant.load [[META0]]
; CHECK-NEXT: [[TMP5:%.*]] = lshr i32 [[TMP2]], 16
; CHECK-NEXT: [[TMP6:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.x()
; CHECK-NEXT: [[TMP7:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.y()
@@ -250,6 +185,5 @@ attributes #0 = { norecurse nounwind "amdgpu-waves-per-eu"="1,1" "amdgpu-flat-wo
attributes #1 = { norecurse nounwind }
;.
; CHECK: [[META0]] = !{}
-; CHECK: [[RNG1]] = !{i32 0, i32 257}
-; CHECK: [[RNG2]] = !{i32 0, i32 1025}
+; CHECK: [[RNG1]] = !{i32 0, i32 1025}
;.
>From 5a6a2d5f9b5fe896ecabb004625aa10b47377ad8 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 29 Sep 2026 19:54:46 +0000
Subject: [PATCH 6/6] Adding a lit-test of bailing-out a memcpy with a phi of
ptrs
---
.../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 11 +-
...promote-alloca-vector-memcpy-phi-select.ll | 104 ++++++++++++++++++
2 files changed, 110 insertions(+), 5 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-vector-memcpy-phi-select.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index b9152458e78503..173f13e5cfccb1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -476,8 +476,8 @@ static bool isSupportedMemset(MemSetInst *I, AllocaInst *AI,
/// Materializes the vector element index that \p Ptr refers to.
///
-/// Recurses through phis and selects, which contribute an index that is itself
-/// a phi or select. All entries are expected to have been created during
+/// Recurses through phis and selects, which contribute an index itself.
+/// All entries of PtrVectorIdx are expected to have been created during
/// analysis, so this only ever updates them and never invalidates the map.
static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
IRBuilder<> B(Ptr->getContext());
@@ -1294,9 +1294,10 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
I->eraseFromParent();
}
- // Delete all the users that are known to be removeable. A loop-carried
- // pointer phi forms a use cycle with the pointers derived from it, so break
- // all the references first rather than relying on a deletion order.
+ // Delete all the users that are known to be removeable.
+ //
+ // UsersToRemove is address-calculating instructions
+ // such as GEP, phi, select, etc.
for (Instruction *I : AA.Vector.UsersToRemove) {
I->dropDroppableUses();
assert(all_of(I->users(),
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-memcpy-phi-select.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-memcpy-phi-select.ll
new file mode 100644
index 00000000000000..6cd1b09b5dfb3f
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-memcpy-phi-select.ll
@@ -0,0 +1,104 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu-amd-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-lds < %s | FileCheck %s
+
+; A mem transfer operand may be a pointer phi or select now that those are part
+; of the alloca's pointer closure, so the offset computation has to bail instead
+; of assuming a getelementptr. Vector promotion needs a constant index for each
+; side of the transfer, and neither of these has one, so the alloca stays put.
+; LDS promotion is disabled so the rejection is visible.
+
+define amdgpu_kernel void @memcpy_phi_dest(i1 %c, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @memcpy_phi_dest(
+; CHECK-SAME: i1 [[C:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; CHECK-NEXT: store i32 7, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT: [[G1:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 4
+; CHECK-NEXT: [[G2:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 8
+; CHECK-NEXT: [[SRC:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 16
+; CHECK-NEXT: br i1 [[C]], label %[[A:.*]], label %[[B:.*]]
+; CHECK: [[A]]:
+; CHECK-NEXT: br label %[[J:.*]]
+; CHECK: [[B]]:
+; CHECK-NEXT: br label %[[J]]
+; CHECK: [[J]]:
+; CHECK-NEXT: [[P:%.*]] = phi ptr addrspace(5) [ [[G1]], %[[A]] ], [ [[G2]], %[[B]] ]
+; CHECK-NEXT: call void @llvm.memcpy.p5.p5.i32(ptr addrspace(5) [[P]], ptr addrspace(5) [[SRC]], i32 8, i1 false)
+; CHECK-NEXT: [[V:%.*]] = load i32, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT: store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [8 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ %g1 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+ %g2 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+ %src = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 16
+ br i1 %c, label %a, label %b
+
+a:
+ br label %j
+
+b:
+ br label %j
+
+j:
+ %p = phi ptr addrspace(5) [ %g1, %a ], [ %g2, %b ]
+ call void @llvm.memcpy.p5.p5.i32(ptr addrspace(5) %p, ptr addrspace(5) %src, i32 8, i1 false)
+ %v = load i32, ptr addrspace(5) %alloca, align 4
+ store i32 %v, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+define amdgpu_kernel void @memcpy_select_dest(i1 %c, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @memcpy_select_dest(
+; CHECK-SAME: i1 [[C:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; CHECK-NEXT: store i32 7, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT: [[G1:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 4
+; CHECK-NEXT: [[G2:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 8
+; CHECK-NEXT: [[SRC:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 16
+; CHECK-NEXT: [[P:%.*]] = select i1 [[C]], ptr addrspace(5) [[G1]], ptr addrspace(5) [[G2]]
+; CHECK-NEXT: call void @llvm.memcpy.p5.p5.i32(ptr addrspace(5) [[P]], ptr addrspace(5) [[SRC]], i32 8, i1 false)
+; CHECK-NEXT: [[V:%.*]] = load i32, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT: store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [8 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ %g1 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+ %g2 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+ %src = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 16
+ %p = select i1 %c, ptr addrspace(5) %g1, ptr addrspace(5) %g2
+ call void @llvm.memcpy.p5.p5.i32(ptr addrspace(5) %p, ptr addrspace(5) %src, i32 8, i1 false)
+ %v = load i32, ptr addrspace(5) %alloca, align 4
+ store i32 %v, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; A transfer between two constant offsets of the same alloca is still promoted,
+; so the bail above is not just rejecting every mem transfer.
+define amdgpu_kernel void @memcpy_const_offsets(ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @memcpy_const_offsets(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA:%.*]] = freeze <8 x i32> poison
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <8 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP0]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 5, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT: store i32 7, ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %alloca = alloca [8 x i32], align 4, addrspace(5)
+ store i32 7, ptr addrspace(5) %alloca, align 4
+ %dst = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+ %src = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 16
+ call void @llvm.memcpy.p5.p5.i32(ptr addrspace(5) %dst, ptr addrspace(5) %src, i32 8, i1 false)
+ %v = load i32, ptr addrspace(5) %alloca, align 4
+ store i32 %v, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+declare void @llvm.memcpy.p5.p5.i32(ptr addrspace(5), ptr addrspace(5), i32, i1)
More information about the llvm-commits
mailing list