[llvm] [AMDGPU][DRAFT] PromoteAllocaToVector - Promo alloca phi (PR #226353)

Yoonseo Choi via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 29 12:57:39 PDT 2026


https://github.com/yoonseoch updated https://github.com/llvm/llvm-project/pull/226353

>From 482fc054202392d5dec921d2daf3f34ae544ef5a Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 22 Sep 2026 01:27:03 +0000
Subject: [PATCH 1/6] Testing relaxing phi with more than 2 incoming values

---
 llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 18 +++++++++++++++++-
 1 file changed, 17 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index aefcabd0d3bf17..f40e918d718e56 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -155,6 +155,9 @@ class AMDGPUPromoteAllocaImpl {
                                        Instruction *UseInst, int OpIdx0,
                                        int OpIdx1) const;
 
+  bool allOpsAreDerivedFromSameAlloca(Value *Alloca, Value *Val,
+                                      Instruction *Inst) const;
+
   /// Check whether we have enough local memory for promotion.
   bool hasSufficientLocalMem(const Function &F);
 
@@ -316,7 +319,8 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
             return RejectUser(Inst, "phi from mixed objects");
           break;
         default:
-          return RejectUser(Inst, "phi with too many operands");
+          if (!allOpsAreDerivedFromSameAlloca(AA.Alloca, Cur, Phi))
+            return RejectUser(Inst, "phi with too many operands");
         }
 
         WorkList.push_back(Inst);
@@ -1380,6 +1384,18 @@ bool AMDGPUPromoteAllocaImpl::binaryOpIsDerivedFromSameAlloca(
   return true;
 }
 
+bool AMDGPUPromoteAllocaImpl::allOpsAreDerivedFromSameAlloca(
+    Value *Alloca, Value *Val, Instruction *Inst) const {
+  for (int i = 0; i < Inst->getNumOperands(); i++) {
+    Value *Op = Inst->getOperand(i);
+    if (Op == Val)
+      continue;
+    if (!binaryOpIsDerivedFromSameAlloca(Alloca, Op, Inst, i, i))
+      return false;
+  }
+  return true;
+}
+
 void AMDGPUPromoteAllocaImpl::analyzePromoteToLDS(AllocaAnalysis &AA) const {
   if (DisablePromoteAllocaToLDS) {
     LLVM_DEBUG(dbgs() << "  Promote alloca to LDS is disabled\n");

>From f240f382f405d2ea7df0d63fe796038bb8296596 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Thu, 24 Sep 2026 06:42:33 +0000
Subject: [PATCH 2/6] Extend promote alloca to vector for phi/select

---
 .../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 208 +++++++++++++-----
 .../AMDGPU/promote-alloca-to-lds-select.ll    |   5 +-
 2 files changed, 156 insertions(+), 57 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index f40e918d718e56..711f90e0762c49 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -29,6 +29,7 @@
 #include "GCNSubtarget.h"
 #include "Utils/AMDGPUBaseInfo.h"
 #include "llvm/ADT/STLExtras.h"
+#include "llvm/ADT/SetVector.h"
 #include "llvm/Analysis/CaptureTracking.h"
 #include "llvm/Analysis/InstSimplifyFolder.h"
 #include "llvm/Analysis/InstructionSimplify.h"
@@ -86,14 +87,20 @@ static cl::opt<unsigned>
                             "when sorting profitable allocas"),
                    cl::init(4));
 
-// We support vector indices of the form ((A * stride) >> shift) + B
-// VarIndex is A, VarMul is stride, VarShift is shift and ConstIndex is B. All
-// parts are optional.
-struct GEPToVectorIndex {
+// We support vector indices of the form ((A * stride) >> shift) + B + index(P)
+// VarIndex is A, VarMul is stride, VarShift is shift, ConstIndex is B and Base
+// is P. All parts are optional.
+//
+// Base is set when the pointer is derived from a phi or select instead of
+// directly from the alloca, in which case its index is computed recursively.
+// Because there is only one variable slot, a pointer with a Base must have no
+// variable offset of its own.
+struct PtrToVectorIndex {
   WeakTrackingVH VarIndex = nullptr; // defaults to 0
   ConstantInt *VarMul = nullptr;     // defaults to 1
   ConstantInt *VarShift = nullptr;   // defaults to 0
   ConstantInt *ConstIndex = nullptr; // defaults to 0
+  Value *Base = nullptr;             // defaults to the alloca (index 0)
   Value *Full = nullptr;
 };
 
@@ -108,12 +115,14 @@ struct AllocaAnalysis {
   DenseSet<Value *> Pointers;
   SmallVector<Use *> Uses;
   unsigned Score = 0;
-  bool HaveSelectOrPHI = false;
   struct {
     FixedVectorType *Ty = nullptr;
     SmallVector<Instruction *> Worklist;
-    SmallVector<Instruction *> UsersToRemove;
-    MapVector<GetElementPtrInst *, GEPToVectorIndex> GEPVectorIdx;
+    // A pointer may be reached through more than one use edge (e.g. a phi whose
+    // incoming values are all derived from the alloca), so this must not hold
+    // duplicates.
+    SetVector<Instruction *> UsersToRemove;
+    MapVector<Value *, PtrToVectorIndex> PtrVectorIdx;
     MapVector<MemTransferInst *, MemTransferInfo> TransferInfo;
   } Vector;
   struct {
@@ -305,7 +314,6 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
         if (!binaryOpIsDerivedFromSameAlloca(AA.Alloca, Cur, SI, 1, 2))
           return RejectUser(Inst, "select from mixed objects");
         WorkList.push_back(Inst);
-        AA.HaveSelectOrPHI = true;
       } else if (auto *Phi = dyn_cast<PHINode>(Inst)) {
         // Repeat for phis.
 
@@ -324,7 +332,6 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
         }
 
         WorkList.push_back(Inst);
-        AA.HaveSelectOrPHI = true;
       }
     }
   }
@@ -467,6 +474,11 @@ static bool isSupportedMemset(MemSetInst *I, AllocaInst *AI,
          match(I->getOperand(2), m_SpecificInt(Size)) && !I->isVolatile();
 }
 
+/// Materializes the vector element index that \p Ptr refers to.
+///
+/// Recurses through phis and selects, which contribute an index that is itself
+/// a phi or select. All entries are expected to have been created during
+/// analysis, so this only ever updates them and never invalidates the map.
 static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
   IRBuilder<> B(Ptr->getContext());
 
@@ -474,43 +486,88 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
   if (Ptr == AA.Alloca)
     return B.getInt32(0);
 
+  auto I = AA.Vector.PtrVectorIdx.find(Ptr);
+  assert(I != AA.Vector.PtrVectorIdx.end() && "Must have entry for pointer!");
+
+  if (Value *Full = I->second.Full)
+    return Full;
+
+  // Phis and selects have no index of their own: theirs is built out of the
+  // indices of their operands.
+  if (auto *Phi = dyn_cast<PHINode>(Ptr)) {
+    // Memoize the new phi before recursing. The pointer phi may be
+    // loop-carried, in which case resolving an incoming value comes back here.
+    PHINode *NewPhi =
+        PHINode::Create(B.getInt32Ty(), Phi->getNumIncomingValues(),
+                        Phi->getName() + ".vecidx", Phi->getIterator());
+    I->second.Full = NewPhi;
+
+    for (unsigned Idx = 0, E = Phi->getNumIncomingValues(); Idx != E; ++Idx) {
+      // SSA guarantees the incoming value dominates the end of the incoming
+      // block, and each index is materialized at its own pointer's definition,
+      // so the index dominates the edge as well.
+      Value *IncomingIdx = calculateVectorIndex(Phi->getIncomingValue(Idx), AA);
+      NewPhi->addIncoming(IncomingIdx, Phi->getIncomingBlock(Idx));
+    }
+    return NewPhi;
+  }
+
+  if (auto *Sel = dyn_cast<SelectInst>(Ptr)) {
+    Value *TrueIdx = calculateVectorIndex(Sel->getTrueValue(), AA);
+    Value *FalseIdx = calculateVectorIndex(Sel->getFalseValue(), AA);
+    B.SetInsertPoint(Sel);
+    Value *Result = B.CreateSelect(Sel->getCondition(), TrueIdx, FalseIdx,
+                                   Sel->getName() + ".vecidx");
+    // Re-find rather than reusing I, which was obtained before recursing.
+    AA.Vector.PtrVectorIdx.find(Ptr)->second.Full = Result;
+    return Result;
+  }
+
   auto *GEP = cast<GetElementPtrInst>(Ptr);
-  auto I = AA.Vector.GEPVectorIdx.find(GEP);
-  assert(I != AA.Vector.GEPVectorIdx.end() && "Must have entry for GEP!");
 
-  if (!I->second.Full) {
-    Value *Result = nullptr;
-    B.SetInsertPoint(GEP);
+  // Resolve the base index first, as this may create instructions and
+  // invalidate iterators into the map.
+  Value *BaseIdx =
+      I->second.Base ? calculateVectorIndex(I->second.Base, AA) : nullptr;
 
-    if (I->second.VarIndex) {
-      Result = I->second.VarIndex;
-      Result = B.CreateSExtOrTrunc(Result, B.getInt32Ty());
+  PtrToVectorIndex &Entry = AA.Vector.PtrVectorIdx.find(Ptr)->second;
+  Value *Result = nullptr;
+  B.SetInsertPoint(GEP);
 
-      if (I->second.VarMul)
-        Result = B.CreateMul(Result, I->second.VarMul);
+  if (Entry.VarIndex) {
+    Result = Entry.VarIndex;
+    Result = B.CreateSExtOrTrunc(Result, B.getInt32Ty());
 
-      if (I->second.VarShift)
-        Result = B.CreateAShr(Result, I->second.VarShift, "", /*isExact*/ true);
-    }
+    if (Entry.VarMul)
+      Result = B.CreateMul(Result, Entry.VarMul);
 
-    if (I->second.ConstIndex) {
-      if (Result)
-        Result = B.CreateAdd(Result, I->second.ConstIndex);
-      else
-        Result = I->second.ConstIndex;
-    }
+    if (Entry.VarShift)
+      Result = B.CreateAShr(Result, Entry.VarShift, "", /*isExact*/ true);
+  }
 
-    if (!Result)
-      Result = B.getInt32(0);
+  if (Entry.ConstIndex) {
+    if (Result)
+      Result = B.CreateAdd(Result, Entry.ConstIndex);
+    else
+      Result = Entry.ConstIndex;
+  }
 
-    I->second.Full = Result;
+  if (BaseIdx) {
+    // A pointer with a Base has no variable offset of its own, so this is at
+    // most a constant added to the base index.
+    assert(!Entry.VarIndex && "Base already occupies the variable slot");
+    Result = Result ? B.CreateAdd(BaseIdx, Result) : BaseIdx;
   }
 
-  return I->second.Full;
+  if (!Result)
+    Result = B.getInt32(0);
+
+  Entry.Full = Result;
+  return Result;
 }
 
-static std::optional<GEPToVectorIndex>
-computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
+static std::optional<PtrToVectorIndex>
+computeGEPToVectorIndex(GetElementPtrInst *GEP, const AllocaAnalysis &AA,
                         Type *VecElemTy, const DataLayout &DL) {
   // TODO: Extracting a "multiple of X" from a GEP might be a useful generic
   // helper.
@@ -545,7 +602,18 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
     CurPtr = CurGEP->getPointerOperand();
   }
 
-  assert(CurPtr == Alloca && "GEP not based on alloca");
+  // The walk ends either at the alloca itself or at a phi/select that merges
+  // pointers into the alloca, whose index is resolved separately.
+  Value *Base = nullptr;
+  if (CurPtr != AA.Alloca) {
+    if (!isa<PHINode, SelectInst>(CurPtr) || !AA.Pointers.contains(CurPtr))
+      return {};
+    // The base index occupies the single variable slot, so this GEP can only
+    // add a constant of its own.
+    if (!VarOffsets.empty())
+      return {};
+    Base = CurPtr;
+  }
 
   int64_t VecElemSize = DL.getTypeAllocSize(VecElemTy);
   if (VarOffsets.size() > 1)
@@ -558,7 +626,8 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
     return {};
   APInt IndexQuot = ConstOffset.sdiv(VecElemSize);
 
-  GEPToVectorIndex Result;
+  PtrToVectorIndex Result;
+  Result.Base = Base;
 
   if (!ConstOffset.isZero())
     Result.ConstIndex = ConstantInt::get(Ctx, IndexQuot.sextOrTrunc(BW));
@@ -981,11 +1050,6 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
 }
 
 void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
-  if (AA.HaveSelectOrPHI) {
-    LLVM_DEBUG(dbgs() << "  Cannot convert to vector due to select or phi\n");
-    return;
-  }
-
   Type *AllocaTy = AA.Alloca->getAllocatedType();
   AA.Vector.Ty = getVectorTypeForAlloca(AllocaTy);
   if (!AA.Vector.Ty)
@@ -1038,12 +1102,32 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
     if (auto *GEP = dyn_cast<GetElementPtrInst>(Inst)) {
       // If we can't compute a vector index from this GEP, then we can't
       // promote this alloca to vector.
-      auto Index = computeGEPToVectorIndex(GEP, AA.Alloca, VecEltTy, DL);
+      auto Index = computeGEPToVectorIndex(GEP, AA, VecEltTy, DL);
       if (!Index)
         return RejectUser(Inst, "cannot compute vector index for GEP");
 
-      AA.Vector.GEPVectorIdx[GEP] = std::move(Index.value());
-      AA.Vector.UsersToRemove.push_back(Inst);
+      AA.Vector.PtrVectorIdx[GEP] = std::move(Index.value());
+      AA.Vector.UsersToRemove.insert(Inst);
+      continue;
+    }
+
+    // A phi or select merging pointers into this alloca becomes a phi or select
+    // of their vector indices. collectAllocaUses has already checked that every
+    // operand is derived from this alloca, but that also admits null, for which
+    // there is no index.
+    if (isa<PHINode, SelectInst>(Inst)) {
+      for (Value *Op : Inst->operands()) {
+        if (!Op->getType()->isPointerTy())
+          continue;
+        Value *Ptr = Op->stripPointerCasts();
+        if (Ptr != AA.Alloca && !AA.Pointers.contains(Ptr))
+          return RejectUser(Inst, "operand is not derived from this alloca");
+      }
+
+      // The index itself is built during promotion; this only reserves the
+      // entry so the commit phase never has to insert into the map.
+      AA.Vector.PtrVectorIdx.insert({Inst, PtrToVectorIndex{}});
+      AA.Vector.UsersToRemove.insert(Inst);
       continue;
     }
 
@@ -1066,9 +1150,11 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
         if (Ptr == AA.Alloca)
           return ConstantInt::get(Ptr->getContext(), APInt(32, 0));
 
-        GetElementPtrInst *GEP = cast<GetElementPtrInst>(Ptr);
-        const auto &GEPI = AA.Vector.GEPVectorIdx.find(GEP)->second;
-        if (GEPI.VarIndex)
+        auto *GEP = dyn_cast<GetElementPtrInst>(Ptr);
+        if (!GEP)
+          return nullptr;
+        const auto &GEPI = AA.Vector.PtrVectorIdx.find(GEP)->second;
+        if (GEPI.VarIndex || GEPI.Base)
           return nullptr;
         if (GEPI.ConstIndex)
           return GEPI.ConstIndex;
@@ -1106,14 +1192,14 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
     if (isAssumeLikeIntrinsic(Inst)) {
       if (!Inst->use_empty())
         return RejectUser(Inst, "assume-like intrinsic cannot have any users");
-      AA.Vector.UsersToRemove.push_back(Inst);
+      AA.Vector.UsersToRemove.insert(Inst);
       continue;
     }
 
     if (isa<ICmpInst>(Inst) && all_of(Inst->users(), [](User *U) {
           return isAssumeLikeIntrinsic(cast<Instruction>(U));
         })) {
-      AA.Vector.UsersToRemove.push_back(Inst);
+      AA.Vector.UsersToRemove.insert(Inst);
       continue;
     }
 
@@ -1202,12 +1288,22 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
     I->eraseFromParent();
   }
 
-  // Delete all the users that are known to be removeable.
-  for (Instruction *I : reverse(AA.Vector.UsersToRemove)) {
+  // Delete all the users that are known to be removeable. A loop-carried
+  // pointer phi forms a use cycle with the pointers derived from it, so break
+  // all the references first rather than relying on a deletion order.
+  for (Instruction *I : AA.Vector.UsersToRemove) {
     I->dropDroppableUses();
-    assert(I->use_empty());
-    I->eraseFromParent();
+    assert(all_of(I->users(),
+                  [&](User *U) {
+                    return AA.Vector.UsersToRemove.contains(
+                        cast<Instruction>(U));
+                  }) &&
+           "Removed instruction still used outside the removed set");
   }
+  for (Instruction *I : AA.Vector.UsersToRemove)
+    I->replaceAllUsesWith(PoisonValue::get(I->getType()));
+  for (Instruction *I : AA.Vector.UsersToRemove)
+    I->eraseFromParent();
 
   // Alloca should now be dead too.
   assert(AA.Alloca->use_empty());
@@ -1386,11 +1482,11 @@ bool AMDGPUPromoteAllocaImpl::binaryOpIsDerivedFromSameAlloca(
 
 bool AMDGPUPromoteAllocaImpl::allOpsAreDerivedFromSameAlloca(
     Value *Alloca, Value *Val, Instruction *Inst) const {
-  for (int i = 0; i < Inst->getNumOperands(); i++) {
-    Value *Op = Inst->getOperand(i);
+  for (unsigned I = 0, E = Inst->getNumOperands(); I != E; ++I) {
+    Value *Op = Inst->getOperand(I);
     if (Op == Val)
       continue;
-    if (!binaryOpIsDerivedFromSameAlloca(Alloca, Op, Inst, i, i))
+    if (!binaryOpIsDerivedFromSameAlloca(Alloca, Op, Inst, I, I))
       return false;
   }
   return true;
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
index b38c2d5b205653..df8d8376cc8dc2 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
@@ -1,5 +1,8 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca < %s | FileCheck %s
+; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-vector < %s | FileCheck %s
+
+; Selects are promotable to vector too, so pin this test to the LDS path.
+; Vector promotion of selects is covered by promote-alloca-vector-select.ll.
 
 define amdgpu_kernel void @lds_promoted_alloca_select_invalid_pointer_operand() #0 {
 ; CHECK-LABEL: define amdgpu_kernel void @lds_promoted_alloca_select_invalid_pointer_operand(

>From 0513170e5ab9f7bd80dd12ec97f8648989370ceb Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Thu, 24 Sep 2026 23:09:27 +0000
Subject: [PATCH 3/6] Clarifications

---
 llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 18 ++++++++++++------
 1 file changed, 12 insertions(+), 6 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index 711f90e0762c49..b9152458e78503 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -328,7 +328,7 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
           break;
         default:
           if (!allOpsAreDerivedFromSameAlloca(AA.Alloca, Cur, Phi))
-            return RejectUser(Inst, "phi with too many operands");
+            return RejectUser(Inst, "phi from mixed objects");
         }
 
         WorkList.push_back(Inst);
@@ -522,7 +522,7 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
     AA.Vector.PtrVectorIdx.find(Ptr)->second.Full = Result;
     return Result;
   }
-
+  // If it was not a phi or select, it must be a GEP.
   auto *GEP = cast<GetElementPtrInst>(Ptr);
 
   // Resolve the base index first, as this may create instructions and
@@ -530,6 +530,7 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
   Value *BaseIdx =
       I->second.Base ? calculateVectorIndex(I->second.Base, AA) : nullptr;
 
+  // Re-find by Ptr to get the most up to date entry.
   PtrToVectorIndex &Entry = AA.Vector.PtrVectorIdx.find(Ptr)->second;
   Value *Result = nullptr;
   B.SetInsertPoint(GEP);
@@ -1116,12 +1117,17 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
     // operand is derived from this alloca, but that also admits null, for which
     // there is no index.
     if (isa<PHINode, SelectInst>(Inst)) {
-      for (Value *Op : Inst->operands()) {
-        if (!Op->getType()->isPointerTy())
-          continue;
+      // Check the merged pointers only. A phi's operands are all incoming
+      // values, but a select's first operand is its condition.
+      auto PtrOps = isa<PHINode>(Inst)
+                        ? cast<PHINode>(Inst)->incoming_values()
+                        : make_range(Inst->op_begin() + 1, Inst->op_end());
+      for (Value *Op : PtrOps) {
+        // collectAllocaUses has already checked that these derive from this
+        // alloca, but it also admits null, which has no vector index.
         Value *Ptr = Op->stripPointerCasts();
         if (Ptr != AA.Alloca && !AA.Pointers.contains(Ptr))
-          return RejectUser(Inst, "operand is not derived from this alloca");
+          return RejectUser(Inst, "operand has no vector index");
       }
 
       // The index itself is built during promotion; this only reserves the

>From 8cbb286b761008484059fddaf4bf32bb14336202 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Fri, 25 Sep 2026 03:51:59 +0000
Subject: [PATCH 4/6] Adding lit-tests

---
 .../promote-alloca-select-vector-preferred.ll | 131 ++++++++
 .../AMDGPU/promote-alloca-vector-phi.ll       | 298 ++++++++++++++++++
 .../AMDGPU/promote-alloca-vector-select.ll    |  96 ++++++
 3 files changed, 525 insertions(+)
 create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll
 create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
 create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-vector-select.ll

diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll
new file mode 100644
index 00000000000000..5def4376976c18
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll
@@ -0,0 +1,131 @@
+; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca < %s | FileCheck %s
+
+; Same functions as promote-alloca-to-lds-select.ll, but with the default
+; pipeline, where promotion to vector is tried before promotion to LDS. Only the
+; allocas that vector promotion turns down fall through to LDS.
+
+; CHECK-LABEL: @lds_promoted_alloca_select_invalid_pointer_operand(
+; CHECK: %alloca = alloca i32
+; CHECK: select i1 undef, ptr addrspace(5) poison, ptr addrspace(5) %alloca
+define amdgpu_kernel void @lds_promoted_alloca_select_invalid_pointer_operand() #0 {
+  %alloca = alloca i32, align 4, addrspace(5)
+  %select = select i1 undef, ptr addrspace(5) poison, ptr addrspace(5) %alloca
+  store i32 0, ptr addrspace(5) %select, align 4
+  ret void
+}
+
+; The select of two pointers becomes a select of the two vector indices.
+; CHECK-LABEL: @lds_promote_alloca_select_two_derived_pointers(
+; CHECK: %alloca = freeze <16 x i32> poison
+; CHECK: %select.vecidx = select i1 undef, i32 %a, i32 %b
+; CHECK: insertelement <16 x i32> %alloca, i32 0, i32 %select.vecidx
+define amdgpu_kernel void @lds_promote_alloca_select_two_derived_pointers(i32 %a, i32 %b) #0 {
+  %alloca = alloca [16 x i32], align 4, addrspace(5)
+  %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
+  %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
+  %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
+  store i32 0, ptr addrspace(5) %select, align 4
+  ret void
+}
+
+; FIXME: This should be promotable but requires knowing that both will be promoted first.
+
+; CHECK-LABEL: @lds_promote_alloca_select_two_allocas(
+; CHECK: %alloca0 = alloca i32, i32 16, align 4
+; CHECK: %alloca1 = alloca i32, i32 16, align 4
+; CHECK: %ptr0 = getelementptr inbounds i32, ptr addrspace(5) %alloca0, i32 %a
+; CHECK: %ptr1 = getelementptr inbounds i32, ptr addrspace(5) %alloca1, i32 %b
+; CHECK: %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
+define amdgpu_kernel void @lds_promote_alloca_select_two_allocas(i32 %a, i32 %b) #0 {
+  %alloca0 = alloca i32, i32 16, align 4, addrspace(5)
+  %alloca1 = alloca i32, i32 16, align 4, addrspace(5)
+  %ptr0 = getelementptr inbounds i32, ptr addrspace(5) %alloca0, i32 %a
+  %ptr1 = getelementptr inbounds i32, ptr addrspace(5) %alloca1, i32 %b
+  %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
+  store i32 0, ptr addrspace(5) %select, align 4
+  ret void
+}
+
+; Both indices are constant, so the select of indices folds away entirely.
+; CHECK-LABEL: @lds_promote_alloca_select_two_derived_constant_pointers(
+; CHECK: %alloca = freeze <16 x i32> poison
+; CHECK-NOT: select
+; CHECK: insertelement <16 x i32> %alloca, i32 0, i32 3
+define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointers() #0 {
+  %alloca = alloca [16 x i32], align 4, addrspace(5)
+  %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 1
+  %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 3
+  %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
+  store i32 0, ptr addrspace(5) %select, align 4
+  ret void
+}
+
+; FIXME: Can be promoted, but we'd have to recursively show that the select
+; operands all point to the same alloca.
+
+; CHECK-LABEL: @lds_promoted_alloca_select_input_select(
+; CHECK: alloca
+define amdgpu_kernel void @lds_promoted_alloca_select_input_select(i32 %a, i32 %b, i32 %c, i1 %c1, i1 %c2) #0 {
+  %alloca = alloca [16 x i32], align 4, addrspace(5)
+  %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
+  %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
+  %ptr2 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %c
+  %select0 = select i1 %c1, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
+  %select1 = select i1 %c2, ptr addrspace(5) %select0, ptr addrspace(5) %ptr2
+  store i32 0, ptr addrspace(5) %select1, align 4
+  ret void
+}
+
+define amdgpu_kernel void @lds_promoted_alloca_select_input_phi(i32 %a, i32 %b, i32 %c, i1 %c0) #0 {
+entry:
+  %alloca = alloca [16 x i32], align 4, addrspace(5)
+  %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
+  %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
+  store i32 0, ptr addrspace(5) %ptr0
+  br i1 %c0, label %bb1, label %bb2
+
+bb1:
+  %ptr2 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %c
+  %select0 = select i1 undef, ptr addrspace(5) poison, ptr addrspace(5) %ptr2
+  store i32 0, ptr addrspace(5) %ptr1
+  br label %bb2
+
+bb2:
+  %phi.ptr = phi ptr addrspace(5) [ %ptr0, %entry ], [ %select0, %bb1 ]
+  %select1 = select i1 undef, ptr addrspace(5) %phi.ptr, ptr addrspace(5) %ptr1
+  store i32 0, ptr addrspace(5) %select1, align 4
+  ret void
+}
+
+; CHECK-LABEL: @select_null_rhs(
+; CHECK-NOT: alloca
+; CHECK: select i1 %tmp2, ptr addrspace(3) %{{[0-9]+}}, ptr addrspace(3) null
+define amdgpu_kernel void @select_null_rhs(ptr addrspace(1) nocapture %arg, i32 %arg1) #1 {
+bb:
+  %tmp = alloca double, align 8, addrspace(5)
+  store double 0.000000e+00, ptr addrspace(5) %tmp, align 8
+  %tmp2 = icmp eq i32 %arg1, 0
+  %tmp3 = select i1 %tmp2, ptr addrspace(5) %tmp, ptr addrspace(5) zeroinitializer
+  store double 1.000000e+00, ptr addrspace(5) %tmp3, align 8
+  %tmp4 = load double, ptr addrspace(5) %tmp, align 8
+  store double %tmp4, ptr addrspace(1) %arg
+  ret void
+}
+
+; CHECK-LABEL: @select_null_lhs(
+; CHECK-NOT: alloca
+; CHECK: select i1 %tmp2, ptr addrspace(3) null, ptr addrspace(3) %{{[0-9]+}}
+define amdgpu_kernel void @select_null_lhs(ptr addrspace(1) nocapture %arg, i32 %arg1) #1 {
+bb:
+  %tmp = alloca double, align 8, addrspace(5)
+  store double 0.000000e+00, ptr addrspace(5) %tmp, align 8
+  %tmp2 = icmp eq i32 %arg1, 0
+  %tmp3 = select i1 %tmp2, ptr addrspace(5) zeroinitializer, ptr addrspace(5) %tmp
+  store double 1.000000e+00, ptr addrspace(5) %tmp3, align 8
+  %tmp4 = load double, ptr addrspace(5) %tmp, align 8
+  store double %tmp4, ptr addrspace(1) %arg
+  ret void
+}
+
+attributes #0 = { norecurse nounwind "amdgpu-waves-per-eu"="1,1" "amdgpu-flat-work-group-size"="1,256" }
+attributes #1 = { norecurse nounwind }
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
new file mode 100644
index 00000000000000..fcfb6253408706
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
@@ -0,0 +1,298 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu-amd-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-lds < %s | FileCheck %s
+
+; A phi merging pointers derived from a single alloca becomes a phi of the
+; corresponding vector indices. LDS promotion is disabled so that allocas which
+; cannot be vectorized are left alone.
+
+; Both incoming values are constant offsets into the alloca, so the index phi is
+; a phi of constants.
+define amdgpu_kernel void @phi_const_geps(i1 %cond, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_const_geps(
+; CHECK-SAME: i1 [[COND:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT:    br i1 [[COND]], label %[[IF:.*]], label %[[ELSE:.*]]
+; CHECK:       [[IF]]:
+; CHECK-NEXT:    br label %[[ENDIF:.*]]
+; CHECK:       [[ELSE]]:
+; CHECK-NEXT:    br label %[[ENDIF]]
+; CHECK:       [[ENDIF]]:
+; CHECK-NEXT:    [[PTR_VECIDX:%.*]] = phi i32 [ 1, %[[IF]] ], [ 2, %[[ELSE]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x i32> [[TMP0]], i32 [[PTR_VECIDX]]
+; CHECK-NEXT:    store i32 [[TMP1]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [4 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  %gep1 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+  %gep2 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+  br i1 %cond, label %if, label %else
+
+if:
+  br label %endif
+
+else:
+  br label %endif
+
+endif:
+  %ptr = phi ptr addrspace(5) [ %gep1, %if ], [ %gep2, %else ]
+  %val = load i32, ptr addrspace(5) %ptr, align 4
+  store i32 %val, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; Three incoming values, one of which is the alloca itself (index 0), with a
+; further constant GEP applied to the phi result. This is the shape SROA's
+; gep(phi) -> phi(gep) fold leaves behind.
+define amdgpu_kernel void @phi_gep_and_alloca_then_gep(i32 %sel, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_gep_and_alloca_then_gep(
+; CHECK-SAME: i32 [[SEL:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT:    switch i32 [[SEL]], label %[[C:.*]] [
+; CHECK-NEXT:      i32 0, label %[[A:.*]]
+; CHECK-NEXT:      i32 1, label %[[B:.*]]
+; CHECK-NEXT:    ]
+; CHECK:       [[A]]:
+; CHECK-NEXT:    br label %[[JOIN:.*]]
+; CHECK:       [[B]]:
+; CHECK-NEXT:    br label %[[JOIN]]
+; CHECK:       [[C]]:
+; CHECK-NEXT:    br label %[[JOIN]]
+; CHECK:       [[JOIN]]:
+; CHECK-NEXT:    [[PTR_VECIDX:%.*]] = phi i32 [ 2, %[[A]] ], [ 2, %[[B]] ], [ 0, %[[C]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[PTR_VECIDX]], 1
+; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <4 x i32> [[TMP0]], i32 [[TMP1]]
+; CHECK-NEXT:    store i32 [[TMP2]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [2 x [2 x i32]], align 16, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  %gep8 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+  switch i32 %sel, label %c [
+  i32 0, label %a
+  i32 1, label %b
+  ]
+
+a:
+  br label %join
+
+b:
+  br label %join
+
+c:
+  br label %join
+
+join:
+  %ptr = phi ptr addrspace(5) [ %gep8, %a ], [ %gep8, %b ], [ %alloca, %c ]
+  %gep4 = getelementptr inbounds i8, ptr addrspace(5) %ptr, i32 4
+  %val = load i32, ptr addrspace(5) %gep4, align 4
+  store i32 %val, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; Variable indices flow through into the index phi.
+define amdgpu_kernel void @phi_var_geps(i1 %cond, i32 %a, i32 %b, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_var_geps(
+; CHECK-SAME: i1 [[COND:%.*]], i32 [[A:%.*]], i32 [[B:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT:    br i1 [[COND]], label %[[IF:.*]], label %[[ELSE:.*]]
+; CHECK:       [[IF]]:
+; CHECK-NEXT:    br label %[[ENDIF:.*]]
+; CHECK:       [[ELSE]]:
+; CHECK-NEXT:    br label %[[ENDIF]]
+; CHECK:       [[ENDIF]]:
+; CHECK-NEXT:    [[PTR_VECIDX:%.*]] = phi i32 [ [[A]], %[[IF]] ], [ [[B]], %[[ELSE]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x i32> [[TMP0]], i32 [[PTR_VECIDX]]
+; CHECK-NEXT:    store i32 [[TMP1]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [4 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  %gepa = getelementptr inbounds [4 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
+  %gepb = getelementptr inbounds [4 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
+  br i1 %cond, label %if, label %else
+
+if:
+  br label %endif
+
+else:
+  br label %endif
+
+endif:
+  %ptr = phi ptr addrspace(5) [ %gepa, %if ], [ %gepb, %else ]
+  %val = load i32, ptr addrspace(5) %ptr, align 4
+  store i32 %val, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; A store through the phi has to write back into the promoted vector.
+define amdgpu_kernel void @phi_store_through_phi(i1 %cond, i32 %val, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_store_through_phi(
+; CHECK-SAME: i1 [[COND:%.*]], i32 [[VAL:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT:    br i1 [[COND]], label %[[IF:.*]], label %[[ELSE:.*]]
+; CHECK:       [[IF]]:
+; CHECK-NEXT:    br label %[[ENDIF:.*]]
+; CHECK:       [[ELSE]]:
+; CHECK-NEXT:    br label %[[ENDIF]]
+; CHECK:       [[ENDIF]]:
+; CHECK-NEXT:    [[PTR_VECIDX:%.*]] = phi i32 [ 1, %[[IF]] ], [ 2, %[[ELSE]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x i32> [[TMP0]], i32 [[VAL]], i32 [[PTR_VECIDX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <4 x i32> [[TMP1]], i32 0
+; CHECK-NEXT:    store i32 [[TMP2]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [4 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  %gep1 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+  %gep2 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+  br i1 %cond, label %if, label %else
+
+if:
+  br label %endif
+
+else:
+  br label %endif
+
+endif:
+  %ptr = phi ptr addrspace(5) [ %gep1, %if ], [ %gep2, %else ]
+  store i32 %val, ptr addrspace(5) %ptr, align 4
+  %reload = load i32, ptr addrspace(5) %alloca, align 4
+  store i32 %reload, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; Negative: the phi mixes two different allocas, so neither is promotable.
+define amdgpu_kernel void @phi_two_allocas(i1 %cond, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_two_allocas(
+; CHECK-SAME: i1 [[COND:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA0:%.*]] = alloca [4 x i32], align 4, addrspace(5)
+; CHECK-NEXT:    [[ALLOCA1:%.*]] = alloca [4 x i32], align 4, addrspace(5)
+; CHECK-NEXT:    store i32 7, ptr addrspace(5) [[ALLOCA0]], align 4
+; CHECK-NEXT:    store i32 9, ptr addrspace(5) [[ALLOCA1]], align 4
+; CHECK-NEXT:    br i1 [[COND]], label %[[IF:.*]], label %[[ELSE:.*]]
+; CHECK:       [[IF]]:
+; CHECK-NEXT:    br label %[[ENDIF:.*]]
+; CHECK:       [[ELSE]]:
+; CHECK-NEXT:    br label %[[ENDIF]]
+; CHECK:       [[ENDIF]]:
+; CHECK-NEXT:    [[PTR:%.*]] = phi ptr addrspace(5) [ [[ALLOCA0]], %[[IF]] ], [ [[ALLOCA1]], %[[ELSE]] ]
+; CHECK-NEXT:    [[VAL:%.*]] = load i32, ptr addrspace(5) [[PTR]], align 4
+; CHECK-NEXT:    store i32 [[VAL]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca0 = alloca [4 x i32], align 4, addrspace(5)
+  %alloca1 = alloca [4 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca0, align 4
+  store i32 9, ptr addrspace(5) %alloca1, align 4
+  br i1 %cond, label %if, label %else
+
+if:
+  br label %endif
+
+else:
+  br label %endif
+
+endif:
+  %ptr = phi ptr addrspace(5) [ %alloca0, %if ], [ %alloca1, %else ]
+  %val = load i32, ptr addrspace(5) %ptr, align 4
+  store i32 %val, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; Negative: a variable-offset GEP applied to the phi would need two variable
+; terms in the index expression, so this must bail rather than assert.
+define amdgpu_kernel void @phi_var_gep_off_phi(i1 %cond, i32 %idx, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_var_gep_off_phi(
+; CHECK-SAME: i1 [[COND:%.*]], i32 [[IDX:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA:%.*]] = alloca [4 x i32], align 4, addrspace(5)
+; CHECK-NEXT:    store i32 7, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT:    [[GEP1:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 4
+; CHECK-NEXT:    [[GEP2:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 8
+; CHECK-NEXT:    br i1 [[COND]], label %[[IF:.*]], label %[[ELSE:.*]]
+; CHECK:       [[IF]]:
+; CHECK-NEXT:    br label %[[ENDIF:.*]]
+; CHECK:       [[ELSE]]:
+; CHECK-NEXT:    br label %[[ENDIF]]
+; CHECK:       [[ENDIF]]:
+; CHECK-NEXT:    [[PTR:%.*]] = phi ptr addrspace(5) [ [[GEP1]], %[[IF]] ], [ [[GEP2]], %[[ELSE]] ]
+; CHECK-NEXT:    [[VGEP:%.*]] = getelementptr inbounds i32, ptr addrspace(5) [[PTR]], i32 [[IDX]]
+; CHECK-NEXT:    [[VAL:%.*]] = load i32, ptr addrspace(5) [[VGEP]], align 4
+; CHECK-NEXT:    store i32 [[VAL]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [4 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  %gep1 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+  %gep2 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+  br i1 %cond, label %if, label %else
+
+if:
+  br label %endif
+
+else:
+  br label %endif
+
+endif:
+  %ptr = phi ptr addrspace(5) [ %gep1, %if ], [ %gep2, %else ]
+  %vgep = getelementptr inbounds i32, ptr addrspace(5) %ptr, i32 %idx
+  %val = load i32, ptr addrspace(5) %vgep, align 4
+  store i32 %val, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; Negative: a loop-carried pointer phi is still rejected up front, because
+; getUnderlyingObject cannot see through the phi to the alloca.
+define amdgpu_kernel void @phi_loop_carried(i32 %n, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_loop_carried(
+; CHECK-SAME: i32 [[N:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[ALLOCA:%.*]] = alloca [4 x i32], align 4, addrspace(5)
+; CHECK-NEXT:    store i32 7, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[PTR:%.*]] = phi ptr addrspace(5) [ [[ALLOCA]], %[[ENTRY]] ], [ [[PTR_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[VAL:%.*]] = load i32, ptr addrspace(5) [[PTR]], align 4
+; CHECK-NEXT:    store i32 [[VAL]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    [[PTR_NEXT]] = getelementptr inbounds i8, ptr addrspace(5) [[PTR]], i32 4
+; CHECK-NEXT:    [[I_NEXT]] = add i32 [[I]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i32 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [4 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  br label %loop
+
+loop:
+  %i = phi i32 [ 0, %entry ], [ %i.next, %loop ]
+  %ptr = phi ptr addrspace(5) [ %alloca, %entry ], [ %ptr.next, %loop ]
+  %val = load i32, ptr addrspace(5) %ptr, align 4
+  store i32 %val, ptr addrspace(1) %out, align 4
+  %ptr.next = getelementptr inbounds i8, ptr addrspace(5) %ptr, i32 4
+  %i.next = add i32 %i, 1
+  %done = icmp eq i32 %i.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-select.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-select.ll
new file mode 100644
index 00000000000000..e9dac6bff51930
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-select.ll
@@ -0,0 +1,96 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu-amd-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-lds < %s | FileCheck %s
+
+; A select between pointers derived from a single alloca becomes a select of the
+; corresponding vector indices. LDS promotion is disabled so that allocas which
+; cannot be vectorized are left alone.
+
+define amdgpu_kernel void @select_const_geps(i1 %cond, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @select_const_geps(
+; CHECK-SAME: i1 [[COND:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT:    [[PTR_VECIDX:%.*]] = select i1 [[COND]], i32 1, i32 2
+; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x i32> [[TMP0]], i32 [[PTR_VECIDX]]
+; CHECK-NEXT:    store i32 [[TMP1]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [4 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  %gep1 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+  %gep2 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+  %ptr = select i1 %cond, ptr addrspace(5) %gep1, ptr addrspace(5) %gep2
+  %val = load i32, ptr addrspace(5) %ptr, align 4
+  store i32 %val, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+define amdgpu_kernel void @select_var_geps(i1 %cond, i32 %a, i32 %b, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @select_var_geps(
+; CHECK-SAME: i1 [[COND:%.*]], i32 [[A:%.*]], i32 [[B:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT:    [[PTR_VECIDX:%.*]] = select i1 [[COND]], i32 [[A]], i32 [[B]]
+; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x i32> [[TMP0]], i32 [[PTR_VECIDX]]
+; CHECK-NEXT:    store i32 [[TMP1]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [4 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  %gepa = getelementptr inbounds [4 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
+  %gepb = getelementptr inbounds [4 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
+  %ptr = select i1 %cond, ptr addrspace(5) %gepa, ptr addrspace(5) %gepb
+  %val = load i32, ptr addrspace(5) %ptr, align 4
+  store i32 %val, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; One side is the alloca itself, and a constant GEP is applied to the result.
+define amdgpu_kernel void @select_alloca_then_gep(i1 %cond, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @select_alloca_then_gep(
+; CHECK-SAME: i1 [[COND:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <4 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT:    [[PTR_VECIDX:%.*]] = select i1 [[COND]], i32 2, i32 0
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[PTR_VECIDX]], 1
+; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <4 x i32> [[TMP0]], i32 [[TMP1]]
+; CHECK-NEXT:    store i32 [[TMP2]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [4 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  %gep8 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+  %ptr = select i1 %cond, ptr addrspace(5) %gep8, ptr addrspace(5) %alloca
+  %gep4 = getelementptr inbounds i8, ptr addrspace(5) %ptr, i32 4
+  %val = load i32, ptr addrspace(5) %gep4, align 4
+  store i32 %val, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; Negative: null has no vector index, so this is not vector promotable even
+; though it is accepted for LDS promotion.
+define amdgpu_kernel void @select_null_operand(i1 %cond, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @select_null_operand(
+; CHECK-SAME: i1 [[COND:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA:%.*]] = alloca [4 x i32], align 4, addrspace(5)
+; CHECK-NEXT:    store i32 7, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT:    [[PTR:%.*]] = select i1 [[COND]], ptr addrspace(5) [[ALLOCA]], ptr addrspace(5) null
+; CHECK-NEXT:    [[VAL:%.*]] = load i32, ptr addrspace(5) [[PTR]], align 4
+; CHECK-NEXT:    store i32 [[VAL]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [4 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  %ptr = select i1 %cond, ptr addrspace(5) %alloca, ptr addrspace(5) null
+  %val = load i32, ptr addrspace(5) %ptr, align 4
+  store i32 %val, ptr addrspace(1) %out, align 4
+  ret void
+}

>From e0c5528832de201b4508c88b0ecd77cd14acd974 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Fri, 25 Sep 2026 17:35:58 +0000
Subject: [PATCH 5/6] Remove tests promotable to vector from
 promote-alloca-to-lds-select.ll

---
 .../promote-alloca-select-vector-preferred.ll | 131 ------------------
 .../AMDGPU/promote-alloca-to-lds-select.ll    |  76 +---------
 2 files changed, 5 insertions(+), 202 deletions(-)
 delete mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll

diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll
deleted file mode 100644
index 5def4376976c18..00000000000000
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-select-vector-preferred.ll
+++ /dev/null
@@ -1,131 +0,0 @@
-; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca < %s | FileCheck %s
-
-; Same functions as promote-alloca-to-lds-select.ll, but with the default
-; pipeline, where promotion to vector is tried before promotion to LDS. Only the
-; allocas that vector promotion turns down fall through to LDS.
-
-; CHECK-LABEL: @lds_promoted_alloca_select_invalid_pointer_operand(
-; CHECK: %alloca = alloca i32
-; CHECK: select i1 undef, ptr addrspace(5) poison, ptr addrspace(5) %alloca
-define amdgpu_kernel void @lds_promoted_alloca_select_invalid_pointer_operand() #0 {
-  %alloca = alloca i32, align 4, addrspace(5)
-  %select = select i1 undef, ptr addrspace(5) poison, ptr addrspace(5) %alloca
-  store i32 0, ptr addrspace(5) %select, align 4
-  ret void
-}
-
-; The select of two pointers becomes a select of the two vector indices.
-; CHECK-LABEL: @lds_promote_alloca_select_two_derived_pointers(
-; CHECK: %alloca = freeze <16 x i32> poison
-; CHECK: %select.vecidx = select i1 undef, i32 %a, i32 %b
-; CHECK: insertelement <16 x i32> %alloca, i32 0, i32 %select.vecidx
-define amdgpu_kernel void @lds_promote_alloca_select_two_derived_pointers(i32 %a, i32 %b) #0 {
-  %alloca = alloca [16 x i32], align 4, addrspace(5)
-  %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
-  %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
-  %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
-  store i32 0, ptr addrspace(5) %select, align 4
-  ret void
-}
-
-; FIXME: This should be promotable but requires knowing that both will be promoted first.
-
-; CHECK-LABEL: @lds_promote_alloca_select_two_allocas(
-; CHECK: %alloca0 = alloca i32, i32 16, align 4
-; CHECK: %alloca1 = alloca i32, i32 16, align 4
-; CHECK: %ptr0 = getelementptr inbounds i32, ptr addrspace(5) %alloca0, i32 %a
-; CHECK: %ptr1 = getelementptr inbounds i32, ptr addrspace(5) %alloca1, i32 %b
-; CHECK: %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
-define amdgpu_kernel void @lds_promote_alloca_select_two_allocas(i32 %a, i32 %b) #0 {
-  %alloca0 = alloca i32, i32 16, align 4, addrspace(5)
-  %alloca1 = alloca i32, i32 16, align 4, addrspace(5)
-  %ptr0 = getelementptr inbounds i32, ptr addrspace(5) %alloca0, i32 %a
-  %ptr1 = getelementptr inbounds i32, ptr addrspace(5) %alloca1, i32 %b
-  %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
-  store i32 0, ptr addrspace(5) %select, align 4
-  ret void
-}
-
-; Both indices are constant, so the select of indices folds away entirely.
-; CHECK-LABEL: @lds_promote_alloca_select_two_derived_constant_pointers(
-; CHECK: %alloca = freeze <16 x i32> poison
-; CHECK-NOT: select
-; CHECK: insertelement <16 x i32> %alloca, i32 0, i32 3
-define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointers() #0 {
-  %alloca = alloca [16 x i32], align 4, addrspace(5)
-  %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 1
-  %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 3
-  %select = select i1 undef, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
-  store i32 0, ptr addrspace(5) %select, align 4
-  ret void
-}
-
-; FIXME: Can be promoted, but we'd have to recursively show that the select
-; operands all point to the same alloca.
-
-; CHECK-LABEL: @lds_promoted_alloca_select_input_select(
-; CHECK: alloca
-define amdgpu_kernel void @lds_promoted_alloca_select_input_select(i32 %a, i32 %b, i32 %c, i1 %c1, i1 %c2) #0 {
-  %alloca = alloca [16 x i32], align 4, addrspace(5)
-  %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
-  %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
-  %ptr2 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %c
-  %select0 = select i1 %c1, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
-  %select1 = select i1 %c2, ptr addrspace(5) %select0, ptr addrspace(5) %ptr2
-  store i32 0, ptr addrspace(5) %select1, align 4
-  ret void
-}
-
-define amdgpu_kernel void @lds_promoted_alloca_select_input_phi(i32 %a, i32 %b, i32 %c, i1 %c0) #0 {
-entry:
-  %alloca = alloca [16 x i32], align 4, addrspace(5)
-  %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
-  %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
-  store i32 0, ptr addrspace(5) %ptr0
-  br i1 %c0, label %bb1, label %bb2
-
-bb1:
-  %ptr2 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %c
-  %select0 = select i1 undef, ptr addrspace(5) poison, ptr addrspace(5) %ptr2
-  store i32 0, ptr addrspace(5) %ptr1
-  br label %bb2
-
-bb2:
-  %phi.ptr = phi ptr addrspace(5) [ %ptr0, %entry ], [ %select0, %bb1 ]
-  %select1 = select i1 undef, ptr addrspace(5) %phi.ptr, ptr addrspace(5) %ptr1
-  store i32 0, ptr addrspace(5) %select1, align 4
-  ret void
-}
-
-; CHECK-LABEL: @select_null_rhs(
-; CHECK-NOT: alloca
-; CHECK: select i1 %tmp2, ptr addrspace(3) %{{[0-9]+}}, ptr addrspace(3) null
-define amdgpu_kernel void @select_null_rhs(ptr addrspace(1) nocapture %arg, i32 %arg1) #1 {
-bb:
-  %tmp = alloca double, align 8, addrspace(5)
-  store double 0.000000e+00, ptr addrspace(5) %tmp, align 8
-  %tmp2 = icmp eq i32 %arg1, 0
-  %tmp3 = select i1 %tmp2, ptr addrspace(5) %tmp, ptr addrspace(5) zeroinitializer
-  store double 1.000000e+00, ptr addrspace(5) %tmp3, align 8
-  %tmp4 = load double, ptr addrspace(5) %tmp, align 8
-  store double %tmp4, ptr addrspace(1) %arg
-  ret void
-}
-
-; CHECK-LABEL: @select_null_lhs(
-; CHECK-NOT: alloca
-; CHECK: select i1 %tmp2, ptr addrspace(3) null, ptr addrspace(3) %{{[0-9]+}}
-define amdgpu_kernel void @select_null_lhs(ptr addrspace(1) nocapture %arg, i32 %arg1) #1 {
-bb:
-  %tmp = alloca double, align 8, addrspace(5)
-  store double 0.000000e+00, ptr addrspace(5) %tmp, align 8
-  %tmp2 = icmp eq i32 %arg1, 0
-  %tmp3 = select i1 %tmp2, ptr addrspace(5) zeroinitializer, ptr addrspace(5) %tmp
-  store double 1.000000e+00, ptr addrspace(5) %tmp3, align 8
-  %tmp4 = load double, ptr addrspace(5) %tmp, align 8
-  store double %tmp4, ptr addrspace(1) %arg
-  ret void
-}
-
-attributes #0 = { norecurse nounwind "amdgpu-waves-per-eu"="1,1" "amdgpu-flat-work-group-size"="1,256" }
-attributes #1 = { norecurse nounwind }
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
index df8d8376cc8dc2..822781c9e38f9c 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-vector < %s | FileCheck %s
+; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca < %s | FileCheck %s
 
 ; Selects are promotable to vector too, so pin this test to the LDS path.
 ; Vector promotion of selects is covered by promote-alloca-vector-select.ll.
@@ -18,38 +18,6 @@ define amdgpu_kernel void @lds_promoted_alloca_select_invalid_pointer_operand()
   ret void
 }
 
-define amdgpu_kernel void @lds_promote_alloca_select_two_derived_pointers(i32 %a, i32 %b) #0 {
-; CHECK-LABEL: define amdgpu_kernel void @lds_promote_alloca_select_two_derived_pointers(
-; CHECK-SAME: i32 [[A:%.*]], i32 [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[TMP1:%.*]] = call noalias nonnull dereferenceable(64) ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 1
-; CHECK-NEXT:    [[TMP3:%.*]] = load i32, ptr addrspace(4) [[TMP2]], align 4, !invariant.load [[META0:![0-9]+]]
-; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 2
-; CHECK-NEXT:    [[TMP5:%.*]] = load i32, ptr addrspace(4) [[TMP4]], align 4, !range [[RNG1:![0-9]+]], !invariant.load [[META0]]
-; CHECK-NEXT:    [[TMP6:%.*]] = lshr i32 [[TMP3]], 16
-; CHECK-NEXT:    [[TMP7:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.x()
-; CHECK-NEXT:    [[TMP8:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.y()
-; CHECK-NEXT:    [[TMP9:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.z()
-; CHECK-NEXT:    [[TMP10:%.*]] = mul nuw nsw i32 [[TMP6]], [[TMP5]]
-; CHECK-NEXT:    [[TMP11:%.*]] = mul i32 [[TMP10]], [[TMP7]]
-; CHECK-NEXT:    [[TMP12:%.*]] = mul nuw nsw i32 [[TMP8]], [[TMP5]]
-; CHECK-NEXT:    [[TMP13:%.*]] = add i32 [[TMP11]], [[TMP12]]
-; CHECK-NEXT:    [[TMP14:%.*]] = add i32 [[TMP13]], [[TMP9]]
-; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr inbounds [256 x [16 x i32]], ptr addrspace(3) @lds_promote_alloca_select_two_derived_pointers.alloca, i32 0, i32 [[TMP14]]
-; CHECK-NEXT:    [[PTR0:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 [[A]]
-; CHECK-NEXT:    [[PTR1:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 [[B]]
-; CHECK-NEXT:    [[SELECT:%.*]] = select i1 poison, ptr addrspace(3) [[PTR0]], ptr addrspace(3) [[PTR1]]
-; CHECK-NEXT:    store i32 0, ptr addrspace(3) [[SELECT]], align 4
-; CHECK-NEXT:    ret void
-;
-  %alloca = alloca [16 x i32], align 4, addrspace(5)
-  %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %a
-  %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 %b
-  %select = select i1 poison, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
-  store i32 0, ptr addrspace(5) %select, align 4
-  ret void
-}
-
 ; FIXME: This should be promotable but requires knowing that both will be promoted first.
 
 define amdgpu_kernel void @lds_promote_alloca_select_two_allocas(i32 %a, i32 %b) #0 {
@@ -72,39 +40,6 @@ define amdgpu_kernel void @lds_promote_alloca_select_two_allocas(i32 %a, i32 %b)
   ret void
 }
 
-; TODO: Maybe this should be canonicalized to select on the constant and GEP after.
-define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointers() #0 {
-; CHECK-LABEL: define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointers(
-; CHECK-SAME: ) #[[ATTR0]] {
-; CHECK-NEXT:    [[TMP1:%.*]] = call noalias nonnull dereferenceable(64) ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 1
-; CHECK-NEXT:    [[TMP3:%.*]] = load i32, ptr addrspace(4) [[TMP2]], align 4, !invariant.load [[META0]]
-; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 2
-; CHECK-NEXT:    [[TMP5:%.*]] = load i32, ptr addrspace(4) [[TMP4]], align 4, !range [[RNG1]], !invariant.load [[META0]]
-; CHECK-NEXT:    [[TMP6:%.*]] = lshr i32 [[TMP3]], 16
-; CHECK-NEXT:    [[TMP7:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.x()
-; CHECK-NEXT:    [[TMP8:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.y()
-; CHECK-NEXT:    [[TMP9:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.z()
-; CHECK-NEXT:    [[TMP10:%.*]] = mul nuw nsw i32 [[TMP6]], [[TMP5]]
-; CHECK-NEXT:    [[TMP11:%.*]] = mul i32 [[TMP10]], [[TMP7]]
-; CHECK-NEXT:    [[TMP12:%.*]] = mul nuw nsw i32 [[TMP8]], [[TMP5]]
-; CHECK-NEXT:    [[TMP13:%.*]] = add i32 [[TMP11]], [[TMP12]]
-; CHECK-NEXT:    [[TMP14:%.*]] = add i32 [[TMP13]], [[TMP9]]
-; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr inbounds [256 x [16 x i32]], ptr addrspace(3) @lds_promote_alloca_select_two_derived_constant_pointers.alloca, i32 0, i32 [[TMP14]]
-; CHECK-NEXT:    [[PTR0:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 1
-; CHECK-NEXT:    [[PTR1:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 3
-; CHECK-NEXT:    [[SELECT:%.*]] = select i1 poison, ptr addrspace(3) [[PTR0]], ptr addrspace(3) [[PTR1]]
-; CHECK-NEXT:    store i32 0, ptr addrspace(3) [[SELECT]], align 4
-; CHECK-NEXT:    ret void
-;
-  %alloca = alloca [16 x i32], align 4, addrspace(5)
-  %ptr0 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 1
-  %ptr1 = getelementptr inbounds [16 x i32], ptr addrspace(5) %alloca, i32 0, i32 3
-  %select = select i1 poison, ptr addrspace(5) %ptr0, ptr addrspace(5) %ptr1
-  store i32 0, ptr addrspace(5) %select, align 4
-  ret void
-}
-
 ; FIXME: Can be promoted, but we'd have to recursively show that the select
 ; operands all point to the same alloca.
 
@@ -176,9 +111,9 @@ define amdgpu_kernel void @select_null_rhs(ptr addrspace(1) nocapture %arg, i32
 ; CHECK-NEXT:  [[BB:.*:]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = call noalias nonnull dereferenceable(64) ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
 ; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 1
-; CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4, !invariant.load [[META0]]
+; CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4, !invariant.load [[META0:![0-9]+]]
 ; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 2
-; CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG2:![0-9]+]], !invariant.load [[META0]]
+; CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG1:![0-9]+]], !invariant.load [[META0]]
 ; CHECK-NEXT:    [[TMP5:%.*]] = lshr i32 [[TMP2]], 16
 ; CHECK-NEXT:    [[TMP6:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.x()
 ; CHECK-NEXT:    [[TMP7:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.y()
@@ -216,7 +151,7 @@ define amdgpu_kernel void @select_null_lhs(ptr addrspace(1) nocapture %arg, i32
 ; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 1
 ; CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4, !invariant.load [[META0]]
 ; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 2
-; CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG2]], !invariant.load [[META0]]
+; CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG1]], !invariant.load [[META0]]
 ; CHECK-NEXT:    [[TMP5:%.*]] = lshr i32 [[TMP2]], 16
 ; CHECK-NEXT:    [[TMP6:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.x()
 ; CHECK-NEXT:    [[TMP7:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.y()
@@ -250,6 +185,5 @@ attributes #0 = { norecurse nounwind "amdgpu-waves-per-eu"="1,1" "amdgpu-flat-wo
 attributes #1 = { norecurse nounwind }
 ;.
 ; CHECK: [[META0]] = !{}
-; CHECK: [[RNG1]] = !{i32 0, i32 257}
-; CHECK: [[RNG2]] = !{i32 0, i32 1025}
+; CHECK: [[RNG1]] = !{i32 0, i32 1025}
 ;.

>From 5a6a2d5f9b5fe896ecabb004625aa10b47377ad8 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 29 Sep 2026 19:54:46 +0000
Subject: [PATCH 6/6] Adding a lit-test of bailing-out a memcpy with a phi of
 ptrs

---
 .../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp |  11 +-
 ...promote-alloca-vector-memcpy-phi-select.ll | 104 ++++++++++++++++++
 2 files changed, 110 insertions(+), 5 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-vector-memcpy-phi-select.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index b9152458e78503..173f13e5cfccb1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -476,8 +476,8 @@ static bool isSupportedMemset(MemSetInst *I, AllocaInst *AI,
 
 /// Materializes the vector element index that \p Ptr refers to.
 ///
-/// Recurses through phis and selects, which contribute an index that is itself
-/// a phi or select. All entries are expected to have been created during
+/// Recurses through phis and selects, which contribute an index itself.
+/// All entries of PtrVectorIdx are expected to have been created during
 /// analysis, so this only ever updates them and never invalidates the map.
 static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
   IRBuilder<> B(Ptr->getContext());
@@ -1294,9 +1294,10 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
     I->eraseFromParent();
   }
 
-  // Delete all the users that are known to be removeable. A loop-carried
-  // pointer phi forms a use cycle with the pointers derived from it, so break
-  // all the references first rather than relying on a deletion order.
+  // Delete all the users that are known to be removeable.
+  //
+  // UsersToRemove is address-calculating instructions
+  // such as GEP, phi, select, etc.
   for (Instruction *I : AA.Vector.UsersToRemove) {
     I->dropDroppableUses();
     assert(all_of(I->users(),
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-memcpy-phi-select.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-memcpy-phi-select.ll
new file mode 100644
index 00000000000000..6cd1b09b5dfb3f
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-memcpy-phi-select.ll
@@ -0,0 +1,104 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu-amd-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-lds < %s | FileCheck %s
+
+; A mem transfer operand may be a pointer phi or select now that those are part
+; of the alloca's pointer closure, so the offset computation has to bail instead
+; of assuming a getelementptr. Vector promotion needs a constant index for each
+; side of the transfer, and neither of these has one, so the alloca stays put.
+; LDS promotion is disabled so the rejection is visible.
+
+define amdgpu_kernel void @memcpy_phi_dest(i1 %c, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @memcpy_phi_dest(
+; CHECK-SAME: i1 [[C:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; CHECK-NEXT:    store i32 7, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT:    [[G1:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 4
+; CHECK-NEXT:    [[G2:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 8
+; CHECK-NEXT:    [[SRC:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 16
+; CHECK-NEXT:    br i1 [[C]], label %[[A:.*]], label %[[B:.*]]
+; CHECK:       [[A]]:
+; CHECK-NEXT:    br label %[[J:.*]]
+; CHECK:       [[B]]:
+; CHECK-NEXT:    br label %[[J]]
+; CHECK:       [[J]]:
+; CHECK-NEXT:    [[P:%.*]] = phi ptr addrspace(5) [ [[G1]], %[[A]] ], [ [[G2]], %[[B]] ]
+; CHECK-NEXT:    call void @llvm.memcpy.p5.p5.i32(ptr addrspace(5) [[P]], ptr addrspace(5) [[SRC]], i32 8, i1 false)
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [8 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  %g1 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+  %g2 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+  %src = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 16
+  br i1 %c, label %a, label %b
+
+a:
+  br label %j
+
+b:
+  br label %j
+
+j:
+  %p = phi ptr addrspace(5) [ %g1, %a ], [ %g2, %b ]
+  call void @llvm.memcpy.p5.p5.i32(ptr addrspace(5) %p, ptr addrspace(5) %src, i32 8, i1 false)
+  %v = load i32, ptr addrspace(5) %alloca, align 4
+  store i32 %v, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+define amdgpu_kernel void @memcpy_select_dest(i1 %c, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @memcpy_select_dest(
+; CHECK-SAME: i1 [[C:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; CHECK-NEXT:    store i32 7, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT:    [[G1:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 4
+; CHECK-NEXT:    [[G2:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 8
+; CHECK-NEXT:    [[SRC:%.*]] = getelementptr inbounds i8, ptr addrspace(5) [[ALLOCA]], i32 16
+; CHECK-NEXT:    [[P:%.*]] = select i1 [[C]], ptr addrspace(5) [[G1]], ptr addrspace(5) [[G2]]
+; CHECK-NEXT:    call void @llvm.memcpy.p5.p5.i32(ptr addrspace(5) [[P]], ptr addrspace(5) [[SRC]], i32 8, i1 false)
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr addrspace(5) [[ALLOCA]], align 4
+; CHECK-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [8 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  %g1 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+  %g2 = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 8
+  %src = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 16
+  %p = select i1 %c, ptr addrspace(5) %g1, ptr addrspace(5) %g2
+  call void @llvm.memcpy.p5.p5.i32(ptr addrspace(5) %p, ptr addrspace(5) %src, i32 8, i1 false)
+  %v = load i32, ptr addrspace(5) %alloca, align 4
+  store i32 %v, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+; A transfer between two constant offsets of the same alloca is still promoted,
+; so the bail above is not just rejecting every mem transfer.
+define amdgpu_kernel void @memcpy_const_offsets(ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @memcpy_const_offsets(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <8 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <8 x i32> [[ALLOCA]], i32 7, i32 0
+; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP0]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 5, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT:    store i32 7, ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %alloca = alloca [8 x i32], align 4, addrspace(5)
+  store i32 7, ptr addrspace(5) %alloca, align 4
+  %dst = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 4
+  %src = getelementptr inbounds i8, ptr addrspace(5) %alloca, i32 16
+  call void @llvm.memcpy.p5.p5.i32(ptr addrspace(5) %dst, ptr addrspace(5) %src, i32 8, i1 false)
+  %v = load i32, ptr addrspace(5) %alloca, align 4
+  store i32 %v, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+declare void @llvm.memcpy.p5.p5.i32(ptr addrspace(5), ptr addrspace(5), i32, i1)



More information about the llvm-commits mailing list