[llvm] [AMDGPU] Promote allocas used by phi/select (PR #221870)

via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 11 00:24:24 PDT 2026


https://github.com/ruiling updated https://github.com/llvm/llvm-project/pull/221870

>From 56907331af61437526f9bbeeaf7acec4b4461ec7 Mon Sep 17 00:00:00 2001
From: Ruiling Song <ruiling.song at amd.com>
Date: Mon, 7 Sep 2026 11:30:42 +0800
Subject: [PATCH 1/3] [AMDGPU] Promote allocas used by phi/select

Standard optimization SROA likes to unfold gep(phi) into phi(gep), see
the related PR #83087 and #80983. Such kind of form would trigger alloca
split in next SROA run. This is bad for AMDGPU as it turns dynamically
indexed alloca into smaller pieces and using PHI to access them.

To aggressively promote this, we can check for compatible allocas that
are being PHI'ed and promote them as one combined vector with each old
alloca being part of the combined vector.

Memory intrinsics operating on them are rejected for now.

Co-Authored-By: Claude
---
 .../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 369 +++++++++++++++---
 .../CodeGen/AMDGPU/frame-index-elimination.ll |   4 +-
 llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.ll   |   4 +-
 .../memory-legalizer-store-infinite-loop.ll   |   2 +-
 .../CodeGen/AMDGPU/promote-alloca-lifetime.ll |   2 +-
 .../CodeGen/AMDGPU/promote-alloca-scoring.ll  |   2 +-
 .../AMDGPU/promote-alloca-to-lds-select.ll    |  65 +--
 .../AMDGPU/promote-alloca-vector-phi.ll       | 236 +++++++++++
 .../AMDGPU/required-export-priority.ll        |   6 +-
 .../AMDGPU/sgpr-scavenge-fi-stack-id.ll       |   2 +-
 10 files changed, 566 insertions(+), 126 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index 249d993ce81e2..aea62360d9536 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -95,6 +95,10 @@ struct GEPToVectorIndex {
   ConstantInt *VarShift = nullptr;   // defaults to 0
   ConstantInt *ConstIndex = nullptr; // defaults to 0
   Value *Full = nullptr;
+  // The root pointer this GEP chain is based on: either a group member alloca
+  // (constant lane) or a pointer phi/select of members (dynamic lane). The
+  // final lane index is (this GEP's within-object offset) + lane(BasePtr).
+  Value *BasePtr = nullptr;
 };
 
 struct MemTransferInfo {
@@ -103,16 +107,42 @@ struct MemTransferInfo {
 };
 
 // Analysis for planning the different strategies of alloca promotion.
+//
+// An AllocaAnalysis represents a *group* of one or more allocas that must be
+// promoted together because their pointers are merged by a phi or select. The
+// common case is a singleton group (Members == {Alloca}). When promoting to
+// vector, the whole group becomes a single combined vector value; each member
+// occupies a contiguous lane range starting at Vector.BaseLane[member].
 struct AllocaAnalysis {
-  AllocaInst *Alloca = nullptr;
+  AllocaInst *Alloca = nullptr; // Primary member (lane 0).
   DenseSet<Value *> Pointers;
   SmallVector<Use *> Uses;
   unsigned Score = 0;
-  bool HaveSelectOrPHI = false;
+  // True if some phi/select merges this alloca's pointer with null or a
+  // non-alloca object. Such merges are still fine for LDS promotion, but they
+  // disable vector grouping (we can't assign the other operand a lane).
+  bool HaveUnpromotableMerge = false;
+
+  // All member allocas of this group, in lane order (Members[0] == Alloca).
+  SmallVector<AllocaInst *> Members;
+  // Other allocas this group is directly linked to via a phi/select of
+  // pointers. Used to form groups; empty once grouping is complete.
+  SmallVector<AllocaInst *> Links;
+
   struct {
     FixedVectorType *Ty = nullptr;
+    // Lane offset of each member alloca within the combined vector.
+    DenseMap<AllocaInst *, unsigned> BaseLane;
+    // Cache of computed lane-index values for pointer phis/selects, so cyclic
+    // pointer phis terminate and are only materialized once.
+    DenseMap<Value *, Value *> IndexCache;
     SmallVector<Instruction *> Worklist;
     SmallVector<Instruction *> UsersToRemove;
+    // Pointer phis/selects that merge group members. Replaced by parallel lane
+    // -index phis/selects and removed (RAUW poison) after promotion. A
+    // SetVector because a merge is a use of multiple members and would
+    // otherwise be recorded once per member.
+    SmallSetVector<Instruction *, 4> PtrMerges;
     MapVector<GetElementPtrInst *, GEPToVectorIndex> GEPVectorIdx;
     MapVector<MemTransferInst *, MemTransferInfo> TransferInfo;
   } Vector;
@@ -121,7 +151,11 @@ struct AllocaAnalysis {
     SmallVector<User *> Worklist;
   } LDS;
 
-  explicit AllocaAnalysis(AllocaInst *Alloca) : Alloca(Alloca) {}
+  explicit AllocaAnalysis(AllocaInst *Alloca) : Alloca(Alloca) {
+    Members.push_back(Alloca);
+  }
+
+  bool isGroup() const { return Members.size() > 1; }
 };
 
 // Shared implementation which can do both promotion to vector and to LDS.
@@ -280,6 +314,51 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
   };
 
   SmallVector<Instruction *, 4> WorkList({AA.Alloca});
+
+  // Classify the "other" operand of a pointer phi/select (i.e. an operand not
+  // derived from Cur). Returns false if the merge makes the alloca entirely
+  // unpromotable (merged with an unknown object). Otherwise records links to
+  // any other allocas (for grouping).
+  //
+  // The operand may itself be a chain of phis/selects (e.g. a phi of a phi), so
+  // we look through them to find the set of underlying root allocas. Visited
+  // guards against cyclic pointer phis.
+  SmallPtrSet<Value *, 8> Visited;
+  std::function<bool(Value *)> ClassifyMergeOperand = [&](Value *Other) -> bool {
+    Other = Other->stripPointerCasts();
+    if (AA.Pointers.contains(Other))
+      return true; // Derived from the same pointer.
+    if (!Visited.insert(Other).second)
+      return true; // Already classified (or on a phi cycle).
+
+    if (isa<ConstantPointerNull, ConstantAggregateZero>(Other)) {
+      // Fine for LDS (null stays null), but we can't give it a vector lane.
+      AA.HaveUnpromotableMerge = true;
+      return true;
+    }
+
+    Value *Obj = getUnderlyingObject(Other);
+    // getUnderlyingObject looks through GEPs/casts but not phi/select; recurse
+    // through those to reach the root allocas.
+    if (auto *Phi = dyn_cast<PHINode>(Obj)) {
+      for (Value *In : Phi->incoming_values())
+        if (!ClassifyMergeOperand(In))
+          return false;
+      return true;
+    }
+    if (auto *SI = dyn_cast<SelectInst>(Obj)) {
+      return ClassifyMergeOperand(SI->getTrueValue()) &&
+             ClassifyMergeOperand(SI->getFalseValue());
+    }
+    auto *OtherAI = dyn_cast<AllocaInst>(Obj);
+    if (!OtherAI)
+      return false;
+
+    if (OtherAI != AA.Alloca)
+      AA.Links.push_back(OtherAI);
+    return true;
+  };
+
   while (!WorkList.empty()) {
     auto *Cur = WorkList.pop_back_val();
     if (find(AA.Pointers, Cur) != AA.Pointers.end())
@@ -297,30 +376,23 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
       if (isa<GetElementPtrInst>(U.getUser())) {
         WorkList.push_back(Inst);
       } else if (auto *SI = dyn_cast<SelectInst>(Inst)) {
-        // Only promote a select if we know that the other select operand is
-        // from another pointer that will also be promoted.
-        if (!binaryOpIsDerivedFromSameAlloca(AA.Alloca, Cur, SI, 1, 2))
-          return RejectUser(Inst, "select from mixed objects");
+        // A select may merge this alloca's pointer with another promotable
+        // alloca (recorded as a link for grouping), with the same alloca, or
+        // with null. Anything else is rejected.
+        if (!ClassifyMergeOperand(SI->getTrueValue()) ||
+            !ClassifyMergeOperand(SI->getFalseValue()))
+          return RejectUser(Inst, "select from incompatible alloca");
         WorkList.push_back(Inst);
-        AA.HaveSelectOrPHI = true;
       } else if (auto *Phi = dyn_cast<PHINode>(Inst)) {
-        // Repeat for phis.
-
-        // TODO: Handle more complex cases. We should be able to replace loops
-        // over arrays.
-        switch (Phi->getNumIncomingValues()) {
-        case 1:
-          break;
-        case 2:
-          if (!binaryOpIsDerivedFromSameAlloca(AA.Alloca, Cur, Phi, 0, 1))
-            return RejectUser(Inst, "phi from mixed objects");
-          break;
-        default:
-          return RejectUser(Inst, "phi with too many operands");
-        }
+        // Repeat for phis. Every incoming value must be derived from a
+        // promotable alloca (this one or a linked one).
+        bool AllOk = true;
+        for (Value *Incoming : Phi->incoming_values())
+          AllOk &= ClassifyMergeOperand(Incoming);
+        if (!AllOk)
+          return RejectUser(Inst, "phi with incompatible alloca");
 
         WorkList.push_back(Inst);
-        AA.HaveSelectOrPHI = true;
       }
     }
   }
@@ -376,28 +448,87 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
       VGPRBudgetRatio;
 
   std::vector<AllocaAnalysis> Allocas;
-  for (Instruction &I : F.getEntryBlock()) {
-    if (AllocaInst *AI = dyn_cast<AllocaInst>(&I)) {
+  {
+    // Collect uses for every candidate alloca independently. This also
+    // records phi/select links between distinct allocas (see collectAllocaUses).
+    std::vector<AllocaAnalysis> Raw;
+    DenseMap<AllocaInst *, int> Index; // -> position in Raw, or -1 if rejected.
+    for (Instruction &I : F.getEntryBlock()) {
+      auto *AI = dyn_cast<AllocaInst>(&I);
       // Array allocations are probably not worth handling, since an allocation
       // of the array type is the canonical form.
-      if (!AI->isStaticAlloca() || AI->isArrayAllocation())
+      if (!AI || !AI->isStaticAlloca() || AI->isArrayAllocation())
         continue;
-
       LLVM_DEBUG(dbgs() << "Analyzing: " << *AI << '\n');
-
       AllocaAnalysis AA{AI};
-      if (collectAllocaUses(AA)) {
-        analyzePromoteToVector(AA);
-        if (PromoteToLDS)
-          analyzePromoteToLDS(AA);
-        if (AA.Vector.Ty || AA.LDS.Enable) {
-          scoreAlloca(AA);
-          Allocas.push_back(std::move(AA));
+      if (!collectAllocaUses(AA)) {
+        Index[AI] = -1;
+        continue;
+      }
+      Index[AI] = Raw.size();
+      Raw.push_back(std::move(AA));
+    }
+
+    for (int I = 0, E = Raw.size(); I != E; ++I) {
+      bool Promotable = true;
+      for (AllocaInst *Linked : Raw[I].Links) {
+        // If this alloca is not promotable, the whole group is not promotable
+        // as well.
+        if (Index[Linked] == -1) {
+          // The whole group can not be promoted
+          Promotable = false;
+          break;
+        }
+      }
+
+      if (Promotable && !Raw[I].Uses.empty()) {
+        for (AllocaInst *Linked : Raw[I].Links) {
+          // We have an linked alloca which appears earlier, this is not leader.
+          //
+          assert(Index[Linked] > I && "This should be leader alloca\n");
+          // Pull from the other member alloca into the current alloca, which
+          // is the leader.
+          AllocaAnalysis &LeaderAA = Raw[I];
+          AllocaAnalysis &MemberAA = Raw[Index[Linked]];
+
+          // Merge the uses into the leader alloca.
+          append_range(LeaderAA.Uses, MemberAA.Uses);
+          // Clear member uses so that it will be skipped later.
+          MemberAA.Uses.clear();
+
+          LeaderAA.Pointers.insert_range(MemberAA.Pointers);
+          LeaderAA.Members.push_back(MemberAA.Alloca);
+          LeaderAA.HaveUnpromotableMerge |= MemberAA.HaveUnpromotableMerge;
         }
+        // We only need to process leader alloca for later steps.
+        Allocas.push_back(Raw[I]);
       }
     }
   }
 
+  // Allocas that are merged by phi/select would have duplicated Use entries,
+  // remove them while preserving order.
+  for (AllocaAnalysis &AA : Allocas) {
+    if (!AA.isGroup())
+      continue;
+    SmallPtrSet<Use *, 16> Seen;
+    llvm::erase_if(AA.Uses, [&](Use *U) { return !Seen.insert(U).second; });
+  }
+
+  for (AllocaAnalysis &AA : Allocas) {
+    analyzePromoteToVector(AA);
+    // LDS promotion is not group-aware; only attempt it for singletons.
+    if (PromoteToLDS && !AA.isGroup())
+      analyzePromoteToLDS(AA);
+  }
+
+  // Drop non-promotable alloca.
+  llvm::erase_if(Allocas, [](const AllocaAnalysis &AA) {
+    return !AA.Vector.Ty && !AA.LDS.Enable;
+  });
+  for (AllocaAnalysis &AA : Allocas)
+    scoreAlloca(AA);
+
   stable_sort(Allocas,
               [](const auto &A, const auto &B) { return A.Score > B.Score; });
 
@@ -413,10 +544,13 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
   SetVector<IntrinsicInst *> DeferredIntrs;
   for (AllocaAnalysis &AA : Allocas) {
     if (AA.Vector.Ty) {
-      std::optional<TypeSize> Size = AA.Alloca->getAllocationSize(DL);
-      assert(Size); // Expected to succeed on non-array alloca.
-      const unsigned AllocaCost = Size->getFixedValue() * 8;
-      // First, check if we have enough budget to vectorize this alloca.
+      unsigned AllocaCost = 0;
+      for (AllocaInst *Member : AA.Members) {
+        std::optional<TypeSize> Size = Member->getAllocationSize(DL);
+        assert(Size); // Expected to succeed on non-array alloca.
+        AllocaCost += Size->getFixedValue() * 8;
+      }
+      // First, check if we have enough budget to vectorize this group.
       if (AllocaCost <= VectorizationBudget) {
         promoteAllocaToVector(AA);
         Changed = true;
@@ -464,17 +598,56 @@ static bool isSupportedMemset(MemSetInst *I, AllocaInst *AI,
 }
 
 static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
-  IRBuilder<> B(Ptr->getContext());
-
+  LLVMContext &Ctx = Ptr->getContext();
   Ptr = Ptr->stripPointerCasts();
-  if (Ptr == AA.Alloca)
-    return B.getInt32(0);
+
+  // Already computed (also breaks cycles for pointer phis).
+  auto Cached = AA.Vector.IndexCache.find(Ptr);
+  if (Cached != AA.Vector.IndexCache.end())
+    return Cached->second;
+
+  const auto BaseLaneConst = [&](AllocaInst *Member) -> Value * {
+    return ConstantInt::get(Type::getInt32Ty(Ctx),
+                            AA.Vector.BaseLane.lookup(Member));
+  };
+
+  // A pointer that is directly one of the member allocas indexes lane
+  // BaseLane[member].
+  if (auto *AI = dyn_cast<AllocaInst>(Ptr)) {
+    Value *Idx = BaseLaneConst(AI);
+    AA.Vector.IndexCache[Ptr] = Idx;
+    return Idx;
+  }
+
+  // Pointer phi: build a parallel phi of lane indices. Insert and cache it
+  // before recursing so self-referential phis terminate.
+  if (auto *Phi = dyn_cast<PHINode>(Ptr)) {
+    IRBuilder<> B(Phi);
+    PHINode *IdxPhi =
+        B.CreatePHI(B.getInt32Ty(), Phi->getNumIncomingValues(), "promotealloca.idx");
+    AA.Vector.IndexCache[Ptr] = IdxPhi;
+    for (unsigned I = 0, E = Phi->getNumIncomingValues(); I != E; ++I)
+      IdxPhi->addIncoming(calculateVectorIndex(Phi->getIncomingValue(I), AA),
+                          Phi->getIncomingBlock(I));
+    return IdxPhi;
+  }
+
+  // Pointer select: select between the two operand lane indices.
+  if (auto *SI = dyn_cast<SelectInst>(Ptr)) {
+    Value *T = calculateVectorIndex(SI->getTrueValue(), AA);
+    Value *F = calculateVectorIndex(SI->getFalseValue(), AA);
+    IRBuilder<> B(SI);
+    Value *Idx = B.CreateSelect(SI->getCondition(), T, F, "promotealloca.idx");
+    AA.Vector.IndexCache[Ptr] = Idx;
+    return Idx;
+  }
 
   auto *GEP = cast<GetElementPtrInst>(Ptr);
   auto I = AA.Vector.GEPVectorIdx.find(GEP);
   assert(I != AA.Vector.GEPVectorIdx.end() && "Must have entry for GEP!");
 
   if (!I->second.Full) {
+    IRBuilder<> B(Ctx);
     Value *Result = nullptr;
     B.SetInsertPoint(GEP);
 
@@ -499,14 +672,24 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
     if (!Result)
       Result = B.getInt32(0);
 
+    // Offset by the lane of the root pointer this GEP is based on. For a member
+    // alloca this is its constant base lane; for a pointer phi/select it is the
+    // parallel lane-index value.
+    if (I->second.BasePtr != AA.Alloca || AA.isGroup()) {
+      Value *BaseIdx = calculateVectorIndex(I->second.BasePtr, AA);
+      if (auto *C = dyn_cast<ConstantInt>(BaseIdx); !C || !C->isZero())
+        Result = B.CreateAdd(Result, BaseIdx);
+    }
+
     I->second.Full = Result;
   }
 
+  AA.Vector.IndexCache[Ptr] = I->second.Full;
   return I->second.Full;
 }
 
 static std::optional<GEPToVectorIndex>
-computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
+computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaAnalysis &AA,
                         Type *VecElemTy, const DataLayout &DL) {
   // TODO: Extracting a "multiple of X" from a GEP might be a useful generic
   // helper.
@@ -541,7 +724,9 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
     CurPtr = CurGEP->getPointerOperand();
   }
 
-  assert(CurPtr == Alloca && "GEP not based on alloca");
+  // This should either points to the alloca or known phi/select.
+  assert(isa<AllocaInst>(CurPtr) ||
+         (isa<PHINode, SelectInst>(CurPtr) /*&& AA.Pointers.contains(CurPtr)*/));
 
   int64_t VecElemSize = DL.getTypeAllocSize(VecElemTy);
   if (VarOffsets.size() > 1)
@@ -555,6 +740,7 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaInst *Alloca,
   APInt IndexQuot = ConstOffset.sdiv(VecElemSize);
 
   GEPToVectorIndex Result;
+  Result.BasePtr = CurPtr;
 
   if (!ConstOffset.isZero())
     Result.ConstIndex = ConstantInt::get(Ctx, IndexQuot.sextOrTrunc(BW));
@@ -903,11 +1089,6 @@ static BasicBlock::iterator skipToNonAllocaInsertPt(BasicBlock &BB,
 
 FixedVectorType *
 AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
-  if (DisablePromoteAllocaToVector) {
-    LLVM_DEBUG(dbgs() << "  Promote alloca to vectors is disabled\n");
-    return nullptr;
-  }
-
   auto *VectorTy = dyn_cast<FixedVectorType>(AllocaTy);
   if (auto *ArrayTy = dyn_cast<ArrayType>(AllocaTy)) {
     uint64_t NumElems = 1;
@@ -966,15 +1147,58 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
 }
 
 void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
-  if (AA.HaveSelectOrPHI) {
-    LLVM_DEBUG(dbgs() << "  Cannot convert to vector due to select or phi\n");
+  // A phi/select that merges with null or a non-alloca object can't be given a
+  // vector lane, so the whole group is unpromotable to vector.
+  if (AA.HaveUnpromotableMerge) {
+    LLVM_DEBUG(dbgs() << "  Cannot convert to vector due to merge with "
+                         "null/unknown object\n");
+    return;
+  }
+
+  if (DisablePromoteAllocaToVector) {
+    LLVM_DEBUG(dbgs() << "  Promote alloca to vectors is disabled\n");
     return;
   }
 
-  Type *AllocaTy = AA.Alloca->getAllocatedType();
-  AA.Vector.Ty = getVectorTypeForAlloca(AllocaTy);
-  if (!AA.Vector.Ty)
+  // Find a valid common element type and assign each member a contiguous lane
+  // range in the combined vector.
+  Type *CombinedEltTy = nullptr;
+  unsigned TotalElems = 0;
+  for (AllocaInst *Member : AA.Members) {
+    Type *MemberTy = Member->getAllocatedType();
+    unsigned ElemCnt;
+    if (MemberTy->isSingleValueType() && !MemberTy->isVectorTy()) {
+      if (CombinedEltTy && CombinedEltTy != MemberTy)
+        return;
+
+      CombinedEltTy = MemberTy;
+      ElemCnt = 1;
+    } else {
+      MemberTy = getVectorTypeForAlloca(MemberTy);
+      // Failed to find a proper vector type or found a different element type,
+      // abort promotion.
+      if (!MemberTy ||
+          (CombinedEltTy && CombinedEltTy != MemberTy->getScalarType()))
+        return;
+      CombinedEltTy = MemberTy->getScalarType();
+      ElemCnt = cast<FixedVectorType>(MemberTy)->getNumElements();
+    }
+
+
+    AA.Vector.BaseLane[Member] = TotalElems;
+    TotalElems += ElemCnt;
+  }
+
+  AA.Vector.Ty = FixedVectorType::get(CombinedEltTy, TotalElems);
+  // Re-check the combined vector against the register-size limit.
+  const unsigned MaxElements =
+      (MaxVectorRegs * 32) / DL.getTypeSizeInBits(CombinedEltTy);
+  if (TotalElems > MaxElements) {
+    LLVM_DEBUG(dbgs() << "  Combined group vector " << *AA.Vector.Ty
+                      << " exceeds the register budget\n");
+    AA.Vector.Ty = nullptr;
     return;
+  }
 
   const auto RejectUser = [&](Instruction *Inst, Twine Msg) {
     LLVM_DEBUG(dbgs() << "  Cannot promote alloca to vector: " << Msg << "\n"
@@ -988,6 +1212,13 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
   for (auto *U : AA.Uses) {
     Instruction *Inst = cast<Instruction>(U->getUser());
 
+    // Pointer phis/selects merging group members are handled by rewriting them
+    // to lane-index phis/selects during promotion.
+    if (isa<PHINode, SelectInst>(Inst) && Inst->getType()->isPointerTy()) {
+      AA.Vector.PtrMerges.insert(Inst);
+      continue;
+    }
+
     if (Value *Ptr = getLoadStorePointerOperand(Inst)) {
       assert(!isa<StoreInst>(Inst) ||
              U->getOperandNo() == StoreInst::getPointerOperandIndex());
@@ -1005,8 +1236,8 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
 
       Ptr = Ptr->stripPointerCasts();
 
-      // Alloca already accessed as vector.
-      if (Ptr == AA.Alloca &&
+      // Alloca already accessed as vector for single alloca group case.
+      if (!AA.isGroup() && Ptr == AA.Alloca &&
           DL.getTypeStoreSize(AA.Alloca->getAllocatedType()) ==
               DL.getTypeStoreSize(AccessTy)) {
         AA.Vector.Worklist.push_back(Inst);
@@ -1023,7 +1254,7 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
     if (auto *GEP = dyn_cast<GetElementPtrInst>(Inst)) {
       // If we can't compute a vector index from this GEP, then we can't
       // promote this alloca to vector.
-      auto Index = computeGEPToVectorIndex(GEP, AA.Alloca, VecEltTy, DL);
+      auto Index = computeGEPToVectorIndex(GEP, AA, VecEltTy, DL);
       if (!Index)
         return RejectUser(Inst, "cannot compute vector index for GEP");
 
@@ -1033,12 +1264,14 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
     }
 
     if (MemSetInst *MSI = dyn_cast<MemSetInst>(Inst);
-        MSI && isSupportedMemset(MSI, AA.Alloca, DL)) {
+        MSI && !AA.isGroup() && isSupportedMemset(MSI, AA.Alloca, DL)) {
       AA.Vector.Worklist.push_back(Inst);
       continue;
     }
 
     if (MemTransferInst *TransferInst = dyn_cast<MemTransferInst>(Inst)) {
+      if (AA.isGroup())
+        return RejectUser(Inst, "mem transfer in grouped alloca not supported");
       if (TransferInst->isVolatile())
         return RejectUser(Inst, "mem transfer inst is volatile");
 
@@ -1187,6 +1420,16 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
     I->eraseFromParent();
   }
 
+  // The pointer phis/selects merging group members are now dead. Drop them
+  // before the GEPs they reference. RAUW with poison first so chained merges
+  // can be erased in any order.
+  for (Instruction *I : AA.Vector.PtrMerges)
+    I->replaceAllUsesWith(PoisonValue::get(I->getType()));
+  for (Instruction *I : AA.Vector.PtrMerges) {
+    assert(I->use_empty());
+    I->eraseFromParent();
+  }
+
   // Delete all the users that are known to be removeable.
   for (Instruction *I : reverse(AA.Vector.UsersToRemove)) {
     I->dropDroppableUses();
@@ -1194,9 +1437,11 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
     I->eraseFromParent();
   }
 
-  // Alloca should now be dead too.
-  assert(AA.Alloca->use_empty());
-  AA.Alloca->eraseFromParent();
+  // All member allocas should now be dead too.
+  for (AllocaInst *Member : AA.Members) {
+    assert(Member->use_empty());
+    Member->eraseFromParent();
+  }
 }
 
 std::pair<Value *, Value *>
diff --git a/llvm/test/CodeGen/AMDGPU/frame-index-elimination.ll b/llvm/test/CodeGen/AMDGPU/frame-index-elimination.ll
index bf4d3ce25a6ef..6f6a5c41cf11c 100644
--- a/llvm/test/CodeGen/AMDGPU/frame-index-elimination.ll
+++ b/llvm/test/CodeGen/AMDGPU/frame-index-elimination.ll
@@ -1,8 +1,8 @@
 ; RUN: llc -mtriple=amdgpu7.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds < %s | FileCheck -enable-var-scope -check-prefixes=GCN,CI,MUBUF %s
 ; RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds < %s | FileCheck -enable-var-scope -check-prefixes=GCN,GFX9,GFX9-MUBUF,MUBUF %s
 ; RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds -mattr=+enable-flat-scratch < %s | FileCheck -enable-var-scope -check-prefixes=GCN,GFX9,GFX9-FLATSCR %s
-; RUN: llc -mtriple=amdgpu11.00-amd-amdhsa -mattr=+real-true16 < %s | FileCheck --check-prefixes=GFX11-TRUE16 %s
-; RUN: llc -mtriple=amdgpu11.00-amd-amdhsa -mattr=-real-true16 < %s | FileCheck --check-prefixes=GFX11-FAKE16 %s
+; RUN: llc -mtriple=amdgpu11.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds -mattr=+real-true16 < %s | FileCheck --check-prefixes=GFX11-TRUE16 %s
+; RUN: llc -mtriple=amdgpu11.00-amd-amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds -mattr=-real-true16 < %s | FileCheck --check-prefixes=GFX11-FAKE16 %s
 
 ; Test that non-entry function frame indices are expanded properly to
 ; give an index relative to the scratch wave offset register
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.ll
index 767aa8228fa0a..59f64a4f6975b 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.ll
@@ -639,8 +639,8 @@ define amdgpu_kernel void @test_export_pos_before_param_across_load(i32 %idx) #0
 ; PREGFX11: {{exp|export}} param0
 ; PREGFX11: {{exp|export}} param1
 define amdgpu_kernel void @test_export_across_store_load(i32 %idx, float %v) #0 {
-  %data0 = alloca <4 x float>, align 8, addrspace(5)
-  %data1 = alloca <4 x float>, align 8, addrspace(5)
+  %data0 = alloca <64 x float>, align 8, addrspace(5)
+  %data1 = alloca <64 x float>, align 8, addrspace(5)
   %cmp = icmp eq i32 %idx, 1
   %data = select i1 %cmp, ptr addrspace(5) %data0, ptr addrspace(5) %data1
   store float %v, ptr addrspace(5) %data, align 8
diff --git a/llvm/test/CodeGen/AMDGPU/memory-legalizer-store-infinite-loop.ll b/llvm/test/CodeGen/AMDGPU/memory-legalizer-store-infinite-loop.ll
index 2786515f1982e..880f8d3972ca1 100644
--- a/llvm/test/CodeGen/AMDGPU/memory-legalizer-store-infinite-loop.ll
+++ b/llvm/test/CodeGen/AMDGPU/memory-legalizer-store-infinite-loop.ll
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -mtriple=amdgpu7.00--amdhsa < %s | FileCheck -check-prefix=GCN %s
+; RUN: llc -mtriple=amdgpu7.00--amdhsa -disable-promote-alloca-to-vector -disable-promote-alloca-to-lds < %s | FileCheck -check-prefix=GCN %s
 
 ; Effectively, check that the compile finishes; in the case
 ; of an infinite loop, llc toggles between merging 2 ST4s
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-lifetime.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-lifetime.ll
index 1dd85cafbaa06..0b0cf9a14acbe 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-lifetime.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-lifetime.ll
@@ -1,4 +1,4 @@
-; RUN: opt -S -mtriple=amdgpu-unknown-amdhsa -passes=amdgpu-promote-alloca %s | FileCheck -check-prefix=OPT %s
+; RUN: opt -S -mtriple=amdgpu-unknown-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-vector %s | FileCheck -check-prefix=OPT %s
 
 declare void @llvm.lifetime.start.p5(i64, ptr addrspace(5) nocapture) #0
 declare void @llvm.lifetime.end.p5(i64, ptr addrspace(5) nocapture) #0
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll
index 09643a826f003..473fa92547a43 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll
@@ -2,10 +2,10 @@
 ; REQUIRES: asserts
 
 ; CHECK-LABEL: Analyzing:   %simpleuser = alloca [4 x i64], align 4, addrspace(5)
+; CHECK-NEXT: Analyzing:   %manyusers = alloca [4 x i64], align 4, addrspace(5)
 ; CHECK-NEXT: Scoring:   %simpleuser = alloca [4 x i64], align 4, addrspace(5)
 ; CHECK-NEXT:   [+1]:   store i64 42, ptr addrspace(5) %simpleuser, align 8
 ; CHECK-NEXT:   => Final Score:1
-; CHECK-LABEL: Analyzing:   %manyusers = alloca [4 x i64], align 4, addrspace(5)
 ; CHECK-NEXT: Scoring:   %manyusers = alloca [4 x i64], align 4, addrspace(5)
 ; CHECK-NEXT:   [+1]:   store i64 %v0.add, ptr addrspace(5) %manyusers.1, align 8
 ; CHECK-NEXT:   [+1]:   %v0 = load i64, ptr addrspace(5) %manyusers.1, align 8
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
index b38c2d5b20565..e7bb3017bc07b 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-to-lds-select.ll
@@ -18,25 +18,9 @@ define amdgpu_kernel void @lds_promoted_alloca_select_invalid_pointer_operand()
 define amdgpu_kernel void @lds_promote_alloca_select_two_derived_pointers(i32 %a, i32 %b) #0 {
 ; CHECK-LABEL: define amdgpu_kernel void @lds_promote_alloca_select_two_derived_pointers(
 ; CHECK-SAME: i32 [[A:%.*]], i32 [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[TMP1:%.*]] = call noalias nonnull dereferenceable(64) ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 1
-; CHECK-NEXT:    [[TMP3:%.*]] = load i32, ptr addrspace(4) [[TMP2]], align 4, !invariant.load [[META0:![0-9]+]]
-; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 2
-; CHECK-NEXT:    [[TMP5:%.*]] = load i32, ptr addrspace(4) [[TMP4]], align 4, !range [[RNG1:![0-9]+]], !invariant.load [[META0]]
-; CHECK-NEXT:    [[TMP6:%.*]] = lshr i32 [[TMP3]], 16
-; CHECK-NEXT:    [[TMP7:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.x()
-; CHECK-NEXT:    [[TMP8:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.y()
-; CHECK-NEXT:    [[TMP9:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.z()
-; CHECK-NEXT:    [[TMP10:%.*]] = mul nuw nsw i32 [[TMP6]], [[TMP5]]
-; CHECK-NEXT:    [[TMP11:%.*]] = mul i32 [[TMP10]], [[TMP7]]
-; CHECK-NEXT:    [[TMP12:%.*]] = mul nuw nsw i32 [[TMP8]], [[TMP5]]
-; CHECK-NEXT:    [[TMP13:%.*]] = add i32 [[TMP11]], [[TMP12]]
-; CHECK-NEXT:    [[TMP14:%.*]] = add i32 [[TMP13]], [[TMP9]]
-; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr inbounds [256 x [16 x i32]], ptr addrspace(3) @lds_promote_alloca_select_two_derived_pointers.alloca, i32 0, i32 [[TMP14]]
-; CHECK-NEXT:    [[PTR0:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 [[A]]
-; CHECK-NEXT:    [[PTR1:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 [[B]]
-; CHECK-NEXT:    [[SELECT:%.*]] = select i1 poison, ptr addrspace(3) [[PTR0]], ptr addrspace(3) [[PTR1]]
-; CHECK-NEXT:    store i32 0, ptr addrspace(3) [[SELECT]], align 4
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <16 x i32> poison
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX:%.*]] = select i1 poison, i32 [[A]], i32 [[B]]
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <16 x i32> [[ALLOCA]], i32 0, i32 [[PROMOTEALLOCA_IDX]]
 ; CHECK-NEXT:    ret void
 ;
   %alloca = alloca [16 x i32], align 4, addrspace(5)
@@ -73,25 +57,7 @@ define amdgpu_kernel void @lds_promote_alloca_select_two_allocas(i32 %a, i32 %b)
 define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointers() #0 {
 ; CHECK-LABEL: define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointers(
 ; CHECK-SAME: ) #[[ATTR0]] {
-; CHECK-NEXT:    [[TMP1:%.*]] = call noalias nonnull dereferenceable(64) ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 1
-; CHECK-NEXT:    [[TMP3:%.*]] = load i32, ptr addrspace(4) [[TMP2]], align 4, !invariant.load [[META0]]
-; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP1]], i64 2
-; CHECK-NEXT:    [[TMP5:%.*]] = load i32, ptr addrspace(4) [[TMP4]], align 4, !range [[RNG1]], !invariant.load [[META0]]
-; CHECK-NEXT:    [[TMP6:%.*]] = lshr i32 [[TMP3]], 16
-; CHECK-NEXT:    [[TMP7:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.x()
-; CHECK-NEXT:    [[TMP8:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.y()
-; CHECK-NEXT:    [[TMP9:%.*]] = call range(i32 0, 256) i32 @llvm.amdgcn.workitem.id.z()
-; CHECK-NEXT:    [[TMP10:%.*]] = mul nuw nsw i32 [[TMP6]], [[TMP5]]
-; CHECK-NEXT:    [[TMP11:%.*]] = mul i32 [[TMP10]], [[TMP7]]
-; CHECK-NEXT:    [[TMP12:%.*]] = mul nuw nsw i32 [[TMP8]], [[TMP5]]
-; CHECK-NEXT:    [[TMP13:%.*]] = add i32 [[TMP11]], [[TMP12]]
-; CHECK-NEXT:    [[TMP14:%.*]] = add i32 [[TMP13]], [[TMP9]]
-; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr inbounds [256 x [16 x i32]], ptr addrspace(3) @lds_promote_alloca_select_two_derived_constant_pointers.alloca, i32 0, i32 [[TMP14]]
-; CHECK-NEXT:    [[PTR0:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 1
-; CHECK-NEXT:    [[PTR1:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(3) [[TMP15]], i32 0, i32 3
-; CHECK-NEXT:    [[SELECT:%.*]] = select i1 poison, ptr addrspace(3) [[PTR0]], ptr addrspace(3) [[PTR1]]
-; CHECK-NEXT:    store i32 0, ptr addrspace(3) [[SELECT]], align 4
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <16 x i32> poison
 ; CHECK-NEXT:    ret void
 ;
   %alloca = alloca [16 x i32], align 4, addrspace(5)
@@ -102,19 +68,13 @@ define amdgpu_kernel void @lds_promote_alloca_select_two_derived_constant_pointe
   ret void
 }
 
-; FIXME: Can be promoted, but we'd have to recursively show that the select
-; operands all point to the same alloca.
-
 define amdgpu_kernel void @lds_promoted_alloca_select_input_select(i32 %a, i32 %b, i32 %c, i1 %c1, i1 %c2) #0 {
 ; CHECK-LABEL: define amdgpu_kernel void @lds_promoted_alloca_select_input_select(
 ; CHECK-SAME: i32 [[A:%.*]], i32 [[B:%.*]], i32 [[C:%.*]], i1 [[C1:%.*]], i1 [[C2:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[ALLOCA:%.*]] = alloca [16 x i32], align 4, addrspace(5)
-; CHECK-NEXT:    [[PTR0:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(5) [[ALLOCA]], i32 0, i32 [[A]]
-; CHECK-NEXT:    [[PTR1:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(5) [[ALLOCA]], i32 0, i32 [[B]]
-; CHECK-NEXT:    [[PTR2:%.*]] = getelementptr inbounds [16 x i32], ptr addrspace(5) [[ALLOCA]], i32 0, i32 [[C]]
-; CHECK-NEXT:    [[SELECT0:%.*]] = select i1 [[C1]], ptr addrspace(5) [[PTR0]], ptr addrspace(5) [[PTR1]]
-; CHECK-NEXT:    [[SELECT1:%.*]] = select i1 [[C2]], ptr addrspace(5) [[SELECT0]], ptr addrspace(5) [[PTR2]]
-; CHECK-NEXT:    store i32 0, ptr addrspace(5) [[SELECT1]], align 4
+; CHECK-NEXT:    [[ALLOCA:%.*]] = freeze <16 x i32> poison
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX:%.*]] = select i1 [[C1]], i32 [[A]], i32 [[B]]
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX1:%.*]] = select i1 [[C2]], i32 [[PROMOTEALLOCA_IDX]], i32 [[C]]
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <16 x i32> [[ALLOCA]], i32 0, i32 [[PROMOTEALLOCA_IDX1]]
 ; CHECK-NEXT:    ret void
 ;
   %alloca = alloca [16 x i32], align 4, addrspace(5)
@@ -173,9 +133,9 @@ define amdgpu_kernel void @select_null_rhs(ptr addrspace(1) nocapture %arg, i32
 ; CHECK-NEXT:  [[BB:.*:]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = call noalias nonnull dereferenceable(64) ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
 ; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 1
-; CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4, !invariant.load [[META0]]
+; CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4, !invariant.load [[META0:![0-9]+]]
 ; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 2
-; CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG2:![0-9]+]], !invariant.load [[META0]]
+; CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG1:![0-9]+]], !invariant.load [[META0]]
 ; CHECK-NEXT:    [[TMP5:%.*]] = lshr i32 [[TMP2]], 16
 ; CHECK-NEXT:    [[TMP6:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.x()
 ; CHECK-NEXT:    [[TMP7:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.y()
@@ -213,7 +173,7 @@ define amdgpu_kernel void @select_null_lhs(ptr addrspace(1) nocapture %arg, i32
 ; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 1
 ; CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr addrspace(4) [[TMP1]], align 4, !invariant.load [[META0]]
 ; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr addrspace(4) [[TMP0]], i64 2
-; CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG2]], !invariant.load [[META0]]
+; CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr addrspace(4) [[TMP3]], align 4, !range [[RNG1]], !invariant.load [[META0]]
 ; CHECK-NEXT:    [[TMP5:%.*]] = lshr i32 [[TMP2]], 16
 ; CHECK-NEXT:    [[TMP6:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.x()
 ; CHECK-NEXT:    [[TMP7:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.y()
@@ -247,6 +207,5 @@ attributes #0 = { norecurse nounwind "amdgpu-waves-per-eu"="1,1" "amdgpu-flat-wo
 attributes #1 = { norecurse nounwind }
 ;.
 ; CHECK: [[META0]] = !{}
-; CHECK: [[RNG1]] = !{i32 0, i32 257}
-; CHECK: [[RNG2]] = !{i32 0, i32 1025}
+; CHECK: [[RNG1]] = !{i32 0, i32 1025}
 ;.
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
new file mode 100644
index 0000000000000..b57874848e497
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
@@ -0,0 +1,236 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu-unknown-amdhsa -passes=amdgpu-promote-alloca-to-vector < %s | FileCheck %s
+
+; Tests for promotion of allocas whose pointers are merged by phis/selects,
+
+define amdgpu_kernel void @phi_two_allocas_basic(i1 %c, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_two_allocas_basic(
+; CHECK-SAME: i1 [[C:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[A:%.*]] = freeze <2 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x i32> [[A]], i32 1, i32 0
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x i32> [[TMP0]], i32 2, i32 1
+; CHECK-NEXT:    br i1 [[C]], label %[[BB1:.*]], label %[[BB2:.*]]
+; CHECK:       [[BB1]]:
+; CHECK-NEXT:    br label %[[JOIN:.*]]
+; CHECK:       [[BB2]]:
+; CHECK-NEXT:    br label %[[JOIN]]
+; CHECK:       [[JOIN]]:
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX:%.*]] = phi i32 [ 0, %[[BB1]] ], [ 1, %[[BB2]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <2 x i32> [[TMP1]], i32 [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT:    store i32 [[TMP2]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %a = alloca i32, addrspace(5)
+  %b = alloca i32, addrspace(5)
+  store i32 1, ptr addrspace(5) %a
+  store i32 2, ptr addrspace(5) %b
+  br i1 %c, label %bb.1, label %bb.2
+bb.1:
+  br label %join
+bb.2:
+  br label %join
+join:
+  %p = phi ptr addrspace(5) [ %a, %bb.1 ], [ %b, %bb.2 ]
+  %l = load i32, ptr addrspace(5) %p
+  store i32 %l, ptr addrspace(1) %out
+  ret void
+}
+
+; GEP uses a (chained) pointer phi: phi-of-phi feeding a gep.
+define amdgpu_kernel void @gep_uses_chained_phi(i1 %c, i1 %d, i32 %v, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @gep_uses_chained_phi(
+; CHECK-SAME: i1 [[C:%.*]], i1 [[D:%.*]], i32 [[V:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[A:%.*]] = freeze <8 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <8 x i32> [[A]], i32 [[V]], i32 0
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <8 x i32> [[TMP0]], i32 [[V]], i32 4
+; CHECK-NEXT:    br i1 [[C]], label %[[BB_1:.*]], label %[[BB_2:.*]]
+; CHECK:       [[BB_1]]:
+; CHECK-NEXT:    br label %[[JOIN1:.*]]
+; CHECK:       [[BB_2]]:
+; CHECK-NEXT:    br label %[[JOIN1]]
+; CHECK:       [[JOIN1]]:
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX1:%.*]] = phi i32 [ 0, %[[BB_1]] ], [ 4, %[[BB_2]] ]
+; CHECK-NEXT:    br i1 [[D]], label %[[BB_3:.*]], label %[[JOIN2:.*]]
+; CHECK:       [[BB_3]]:
+; CHECK-NEXT:    br label %[[JOIN2]]
+; CHECK:       [[JOIN2]]:
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX:%.*]] = phi i32 [ [[PROMOTEALLOCA_IDX1]], %[[JOIN1]] ], [ 0, %[[BB_3]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add i32 1, [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <8 x i32> [[TMP1]], i32 [[TMP2]]
+; CHECK-NEXT:    store i32 [[TMP3]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %a = alloca [4 x i32], addrspace(5)
+  %b = alloca [4 x i32], addrspace(5)
+  store i32 %v, ptr addrspace(5) %a
+  store i32 %v, ptr addrspace(5) %b
+  br i1 %c, label %bb.1, label %bb.2
+bb.1:
+  br label %join1
+bb.2:
+  br label %join1
+join1:
+  %p1 = phi ptr addrspace(5) [ %a, %bb.1 ], [ %b, %bb.2 ]
+  br i1 %d, label %bb.3, label %join2
+bb.3:
+  br label %join2
+join2:
+  %p2 = phi ptr addrspace(5) [ %p1, %join1 ], [ %a, %bb.3 ]
+  %g = getelementptr i32, ptr addrspace(5) %p2, i32 1
+  %l = load i32, ptr addrspace(5) %g
+  store i32 %l, ptr addrspace(1) %out
+  ret void
+}
+
+; Pointer phi whose incoming values are GEPs (phi uses gep), and a gep uses the
+; phi. Exercises both erase-order directions in a single group.
+define amdgpu_kernel void @merge_uses_gep(i1 %c, i32 %v, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @merge_uses_gep(
+; CHECK-SAME: i1 [[C:%.*]], i32 [[V:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[B:%.*]] = freeze <8 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <8 x i32> [[B]], i32 [[V]], i32 2
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <8 x i32> [[TMP0]], i32 [[V]], i32 7
+; CHECK-NEXT:    br i1 [[C]], label %[[BB1:.*]], label %[[BB2:.*]]
+; CHECK:       [[BB1]]:
+; CHECK-NEXT:    br label %[[JOIN:.*]]
+; CHECK:       [[BB2]]:
+; CHECK-NEXT:    br label %[[JOIN]]
+; CHECK:       [[JOIN]]:
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX:%.*]] = phi i32 [ 2, %[[BB1]] ], [ 7, %[[BB2]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add i32 1, [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <8 x i32> [[TMP1]], i32 [[TMP2]]
+; CHECK-NEXT:    store i32 [[TMP3]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %a = alloca [4 x i32], addrspace(5)
+  %b = alloca [4 x i32], addrspace(5)
+  %ga = getelementptr i32, ptr addrspace(5) %a, i32 2
+  %gb = getelementptr i32, ptr addrspace(5) %b, i32 3
+  store i32 %v, ptr addrspace(5) %ga
+  store i32 %v, ptr addrspace(5) %gb
+  br i1 %c, label %bb.1, label %bb.2
+bb.1:
+  br label %join
+bb.2:
+  br label %join
+join:
+  %p = phi ptr addrspace(5) [ %ga, %bb.1 ], [ %gb, %bb.2 ]
+  %g2 = getelementptr i32, ptr addrspace(5) %p, i32 1
+  %l = load i32, ptr addrspace(5) %g2
+  store i32 %l, ptr addrspace(1) %out
+  ret void
+}
+
+; Chained selects: select of a select.
+define amdgpu_kernel void @select_chain(i1 %c, i1 %d, i32 %v, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @select_chain(
+; CHECK-SAME: i1 [[C:%.*]], i1 [[D:%.*]], i32 [[V:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[B:%.*]] = freeze <8 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <8 x i32> [[B]], i32 [[V]], i32 0
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <8 x i32> [[TMP0]], i32 [[V]], i32 4
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX:%.*]] = select i1 [[C]], i32 0, i32 4
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX1:%.*]] = select i1 [[D]], i32 [[PROMOTEALLOCA_IDX]], i32 0
+; CHECK-NEXT:    [[TMP2:%.*]] = add i32 1, [[PROMOTEALLOCA_IDX1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <8 x i32> [[TMP1]], i32 [[V]], i32 [[TMP2]]
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <8 x i32> [[TMP3]], i32 [[PROMOTEALLOCA_IDX1]]
+; CHECK-NEXT:    store i32 [[TMP4]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %a = alloca [4 x i32], addrspace(5)
+  %b = alloca [4 x i32], addrspace(5)
+  store i32 %v, ptr addrspace(5) %a
+  store i32 %v, ptr addrspace(5) %b
+  %s1 = select i1 %c, ptr addrspace(5) %a, ptr addrspace(5) %b
+  %s2 = select i1 %d, ptr addrspace(5) %s1, ptr addrspace(5) %a
+  %g = getelementptr i32, ptr addrspace(5) %s2, i32 1
+  store i32 %v, ptr addrspace(5) %g
+  %l = load i32, ptr addrspace(5) %s2
+  store i32 %l, ptr addrspace(1) %out
+  ret void
+}
+
+; Self-referential pointer phi across a loop back-edge.
+define amdgpu_kernel void @loop_backedge_phi(i32 %n, i32 %v, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @loop_backedge_phi(
+; CHECK-SAME: i32 [[N:%.*]], i32 [[V:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[A:%.*]] = freeze <8 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <8 x i32> [[A]], i32 [[V]], i32 0
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[TMP2:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <8 x i32> [[TMP0]], i32 [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT:    store i32 [[TMP1]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    [[TMP2]] = add i32 1, [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT:    [[I_NEXT]] = add i32 [[I]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %a = alloca [8 x i32], addrspace(5)
+  store i32 %v, ptr addrspace(5) %a
+  br label %loop
+loop:
+  %i = phi i32 [ 0, %entry ], [ %i.next, %loop ]
+  %p = phi ptr addrspace(5) [ %a, %entry ], [ %p.next, %loop ]
+  %l = load i32, ptr addrspace(5) %p
+  store i32 %l, ptr addrspace(1) %out
+  %p.next = getelementptr i32, ptr addrspace(5) %p, i32 1
+  %i.next = add i32 %i, 1
+  %cmp = icmp slt i32 %i.next, %n
+  br i1 %cmp, label %loop, label %exit
+exit:
+  ret void
+}
+
+; Group members of different lengths combine into one wider vector:
+; [4 x i32] + [2 x i32] -> <6 x i32>, with each member at a contiguous lane range.
+define amdgpu_kernel void @phi_diff_len(i1 %c, i32 %v, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_diff_len(
+; CHECK-SAME: i1 [[C:%.*]], i32 [[V:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[B:%.*]] = freeze <6 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <6 x i32> [[B]], i32 [[V]], i32 0
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <6 x i32> [[TMP0]], i32 [[V]], i32 4
+; CHECK-NEXT:    br i1 [[C]], label %[[BB1:.*]], label %[[BB2:.*]]
+; CHECK:       [[BB1]]:
+; CHECK-NEXT:    br label %[[JOIN:.*]]
+; CHECK:       [[BB2]]:
+; CHECK-NEXT:    br label %[[JOIN]]
+; CHECK:       [[JOIN]]:
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX:%.*]] = phi i32 [ 0, %[[BB1]] ], [ 4, %[[BB2]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add i32 1, [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <6 x i32> [[TMP1]], i32 [[V]], i32 [[TMP2]]
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <6 x i32> [[TMP3]], i32 [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT:    store i32 [[TMP4]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %a = alloca [4 x i32], addrspace(5)
+  %b = alloca [2 x i32], addrspace(5)
+  store i32 %v, ptr addrspace(5) %a
+  store i32 %v, ptr addrspace(5) %b
+  br i1 %c, label %bb.1, label %bb.2
+bb.1:
+  br label %join
+bb.2:
+  br label %join
+join:
+  %p = phi ptr addrspace(5) [ %a, %bb.1 ], [ %b, %bb.2 ]
+  %g = getelementptr i32, ptr addrspace(5) %p, i32 1
+  store i32 %v, ptr addrspace(5) %g
+  %l = load i32, ptr addrspace(5) %p
+  store i32 %l, ptr addrspace(1) %out
+  ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/required-export-priority.ll b/llvm/test/CodeGen/AMDGPU/required-export-priority.ll
index b6e2e75c9d28c..4df215a3f6224 100644
--- a/llvm/test/CodeGen/AMDGPU/required-export-priority.ll
+++ b/llvm/test/CodeGen/AMDGPU/required-export-priority.ll
@@ -1,7 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -mtriple=amdgpu11.00 -amdgpu-enable-vopd=0 < %s | FileCheck -check-prefixes=GCN,GFX11 %s
-; RUN: llc -mtriple=amdgpu11.50 -amdgpu-enable-vopd=0 < %s | FileCheck -check-prefixes=GCN,GFX1150 %s
-; RUN: llc -mtriple=amdgpu11.70 -amdgpu-enable-vopd=0 < %s | FileCheck -check-prefixes=GCN,GFX11 %s
+; RUN: llc -mtriple=amdgpu11.00 -amdgpu-enable-vopd=0 -disable-promote-alloca-to-vector < %s | FileCheck -check-prefixes=GCN,GFX11 %s
+; RUN: llc -mtriple=amdgpu11.50 -amdgpu-enable-vopd=0 -disable-promote-alloca-to-vector < %s | FileCheck -check-prefixes=GCN,GFX1150 %s
+; RUN: llc -mtriple=amdgpu11.70 -amdgpu-enable-vopd=0 -disable-promote-alloca-to-vector < %s | FileCheck -check-prefixes=GCN,GFX11 %s
 
 define amdgpu_ps void @test_export_zeroes_f32() #0 {
 ; GFX11-LABEL: test_export_zeroes_f32:
diff --git a/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll b/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll
index 4d33d0020fff0..9b523a8d7b97f 100644
--- a/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll
+++ b/llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -O1 -mtriple=amdgpu9.0a-amd-amdhsa < %s | FileCheck %s
+; RUN: llc -O1 -mtriple=amdgpu9.0a-amd-amdhsa -disable-promote-alloca-to-vector < %s | FileCheck %s
 
 define void @sgpr_scavenge_fi_stack_id(double %input, i1 %enter_fma_path, i1 %repeat_outer_loop, i1 %enter_sgpr_loop, i1 %enter_inner_loop, i1 %exit_inner_loop, i1 %zero_fma_result) {
 ; CHECK-LABEL: sgpr_scavenge_fi_stack_id:

>From a3c74fa62b607d0fa551e3f6ca01870cd5959206 Mon Sep 17 00:00:00 2001
From: Ruiling Song <ruiling.song at amd.com>
Date: Thu, 10 Sep 2026 16:58:35 +0800
Subject: [PATCH 2/3] Address review comments:

1. depends on getUnderlyingObjects to collect merged alloca
2. Refine the Link propagation based on a Map and add a test for link
   propagation.
3. Change `Uses` to unordered set to handle duplication easily.
4. Fix UTC tests issues.
5. Change `UsersToRemove` to SmallSetVector to handle duplicated PHIs.
---
 .../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 302 +++++++-----------
 .../CodeGen/AMDGPU/promote-alloca-scoring.ll  |  27 +-
 .../AMDGPU/promote-alloca-vector-phi.ll       |  46 ++-
 .../Inputs/amdgpu_asm.ll                      |   2 +-
 .../Inputs/amdgpu_asm.ll.expected             |   2 +-
 .../Inputs/amdgpu_generated_funcs.ll          |   2 +-
 ...dgpu_generated_funcs.ll.generated.expected |   2 +-
 ...pu_generated_funcs.ll.nogenerated.expected |   2 +-
 .../Inputs/amdgpu_isel.ll                     |   2 +-
 .../Inputs/amdgpu_isel.ll.expected            |   2 +-
 10 files changed, 187 insertions(+), 202 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index aea62360d9536..7299ca3ffc3ea 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -88,16 +88,15 @@ static cl::opt<unsigned>
 
 // We support vector indices of the form ((A * stride) >> shift) + B
 // VarIndex is A, VarMul is stride, VarShift is shift and ConstIndex is B. All
-// parts are optional.
+// parts are optional. When BasePtr does not point to Alloca object, the final
+// index will be (index_of_BasePtr + index_of_gep).
 struct GEPToVectorIndex {
   WeakTrackingVH VarIndex = nullptr; // defaults to 0
   ConstantInt *VarMul = nullptr;     // defaults to 1
   ConstantInt *VarShift = nullptr;   // defaults to 0
   ConstantInt *ConstIndex = nullptr; // defaults to 0
   Value *Full = nullptr;
-  // The root pointer this GEP chain is based on: either a group member alloca
-  // (constant lane) or a pointer phi/select of members (dynamic lane). The
-  // final lane index is (this GEP's within-object offset) + lane(BasePtr).
+  // The root pointer this GEP chain is based on.
   Value *BasePtr = nullptr;
 };
 
@@ -108,19 +107,16 @@ struct MemTransferInfo {
 
 // Analysis for planning the different strategies of alloca promotion.
 //
-// An AllocaAnalysis represents a *group* of one or more allocas that must be
-// promoted together because their pointers are merged by a phi or select. The
-// common case is a singleton group (Members == {Alloca}). When promoting to
-// vector, the whole group becomes a single combined vector value; each member
-// occupies a contiguous lane range starting at Vector.BaseLane[member].
+// An AllocaAnalysis may represent one alloca or a group of allocas if they need
+// to be promoted together.
 struct AllocaAnalysis {
   AllocaInst *Alloca = nullptr; // Primary member (lane 0).
   DenseSet<Value *> Pointers;
-  SmallVector<Use *> Uses;
+  SmallDenseSet<Use *> Uses;
   unsigned Score = 0;
   // True if some phi/select merges this alloca's pointer with null or a
   // non-alloca object. Such merges are still fine for LDS promotion, but they
-  // disable vector grouping (we can't assign the other operand a lane).
+  // disable vector promotion.
   bool HaveUnpromotableMerge = false;
 
   // All member allocas of this group, in lane order (Members[0] == Alloca).
@@ -137,12 +133,7 @@ struct AllocaAnalysis {
     // pointer phis terminate and are only materialized once.
     DenseMap<Value *, Value *> IndexCache;
     SmallVector<Instruction *> Worklist;
-    SmallVector<Instruction *> UsersToRemove;
-    // Pointer phis/selects that merge group members. Replaced by parallel lane
-    // -index phis/selects and removed (RAUW poison) after promotion. A
-    // SetVector because a merge is a use of multiple members and would
-    // otherwise be recorded once per member.
-    SmallSetVector<Instruction *, 4> PtrMerges;
+    SmallSetVector<Instruction *, 8> UsersToRemove;
     MapVector<GetElementPtrInst *, GEPToVectorIndex> GEPVectorIdx;
     MapVector<MemTransferInst *, MemTransferInfo> TransferInfo;
   } Vector;
@@ -315,55 +306,14 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
 
   SmallVector<Instruction *, 4> WorkList({AA.Alloca});
 
-  // Classify the "other" operand of a pointer phi/select (i.e. an operand not
-  // derived from Cur). Returns false if the merge makes the alloca entirely
-  // unpromotable (merged with an unknown object). Otherwise records links to
-  // any other allocas (for grouping).
-  //
-  // The operand may itself be a chain of phis/selects (e.g. a phi of a phi), so
-  // we look through them to find the set of underlying root allocas. Visited
-  // guards against cyclic pointer phis.
   SmallPtrSet<Value *, 8> Visited;
-  std::function<bool(Value *)> ClassifyMergeOperand = [&](Value *Other) -> bool {
-    Other = Other->stripPointerCasts();
-    if (AA.Pointers.contains(Other))
-      return true; // Derived from the same pointer.
-    if (!Visited.insert(Other).second)
-      return true; // Already classified (or on a phi cycle).
-
-    if (isa<ConstantPointerNull, ConstantAggregateZero>(Other)) {
-      // Fine for LDS (null stays null), but we can't give it a vector lane.
-      AA.HaveUnpromotableMerge = true;
-      return true;
-    }
-
-    Value *Obj = getUnderlyingObject(Other);
-    // getUnderlyingObject looks through GEPs/casts but not phi/select; recurse
-    // through those to reach the root allocas.
-    if (auto *Phi = dyn_cast<PHINode>(Obj)) {
-      for (Value *In : Phi->incoming_values())
-        if (!ClassifyMergeOperand(In))
-          return false;
-      return true;
-    }
-    if (auto *SI = dyn_cast<SelectInst>(Obj)) {
-      return ClassifyMergeOperand(SI->getTrueValue()) &&
-             ClassifyMergeOperand(SI->getFalseValue());
-    }
-    auto *OtherAI = dyn_cast<AllocaInst>(Obj);
-    if (!OtherAI)
-      return false;
-
-    if (OtherAI != AA.Alloca)
-      AA.Links.push_back(OtherAI);
-    return true;
-  };
-
+  // Other allocas which are merged by phi/select with current alloca.
+  SmallSetVector<AllocaInst *, 4> MergedAllocas;
   while (!WorkList.empty()) {
     auto *Cur = WorkList.pop_back_val();
-    if (find(AA.Pointers, Cur) != AA.Pointers.end())
+    if (!Visited.insert(Cur).second)
       continue;
-    AA.Pointers.insert(Cur);
+
     for (auto &U : Cur->uses()) {
       auto *Inst = cast<Instruction>(U.getUser());
       if (isa<StoreInst>(Inst)) {
@@ -371,31 +321,32 @@ bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &AA) const {
           return RejectUser(Inst, "pointer escapes via store");
         }
       }
-      AA.Uses.push_back(&U);
+      AA.Uses.insert(&U);
 
       if (isa<GetElementPtrInst>(U.getUser())) {
         WorkList.push_back(Inst);
-      } else if (auto *SI = dyn_cast<SelectInst>(Inst)) {
-        // A select may merge this alloca's pointer with another promotable
-        // alloca (recorded as a link for grouping), with the same alloca, or
-        // with null. Anything else is rejected.
-        if (!ClassifyMergeOperand(SI->getTrueValue()) ||
-            !ClassifyMergeOperand(SI->getFalseValue()))
-          return RejectUser(Inst, "select from incompatible alloca");
-        WorkList.push_back(Inst);
-      } else if (auto *Phi = dyn_cast<PHINode>(Inst)) {
-        // Repeat for phis. Every incoming value must be derived from a
-        // promotable alloca (this one or a linked one).
-        bool AllOk = true;
-        for (Value *Incoming : Phi->incoming_values())
-          AllOk &= ClassifyMergeOperand(Incoming);
-        if (!AllOk)
-          return RejectUser(Inst, "phi with incompatible alloca");
-
-        WorkList.push_back(Inst);
+      } else if (isa<SelectInst, PHINode>(Inst)) {
+        SmallVector<const Value *> BaseObjs;
+        getUnderlyingObjects(Inst, BaseObjs, &LI);
+
+        for (auto *Obj : BaseObjs) {
+          if (auto *BaseAlloca = dyn_cast<AllocaInst>(Obj)) {
+            if (BaseAlloca != AA.Alloca) {
+              MergedAllocas.insert(const_cast<AllocaInst *>(BaseAlloca));
+            }
+            WorkList.push_back(Inst);
+          } else if (isa<ConstantPointerNull, ConstantAggregateZero>(Obj)) {
+            AA.HaveUnpromotableMerge = true;
+            WorkList.push_back(Inst);
+          } else {
+            return RejectUser(Inst, "phi/select with unkown object");
+          }
+        }
       }
     }
   }
+  for (auto *Alloca : MergedAllocas)
+    AA.Links.push_back(Alloca);
   return true;
 }
 
@@ -447,72 +398,72 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
                                   : (MaxVGPRs * 32)) /
       VGPRBudgetRatio;
 
-  std::vector<AllocaAnalysis> Allocas;
-  {
-    // Collect uses for every candidate alloca independently. This also
-    // records phi/select links between distinct allocas (see collectAllocaUses).
-    std::vector<AllocaAnalysis> Raw;
-    DenseMap<AllocaInst *, int> Index; // -> position in Raw, or -1 if rejected.
-    for (Instruction &I : F.getEntryBlock()) {
-      auto *AI = dyn_cast<AllocaInst>(&I);
-      // Array allocations are probably not worth handling, since an allocation
-      // of the array type is the canonical form.
-      if (!AI || !AI->isStaticAlloca() || AI->isArrayAllocation())
-        continue;
-      LLVM_DEBUG(dbgs() << "Analyzing: " << *AI << '\n');
-      AllocaAnalysis AA{AI};
-      if (!collectAllocaUses(AA)) {
-        Index[AI] = -1;
-        continue;
-      }
-      Index[AI] = Raw.size();
-      Raw.push_back(std::move(AA));
-    }
+  SmallMapVector<AllocaInst *, AllocaAnalysis, 4> AllocaAnalysisMap;
+  for (Instruction &I : F.getEntryBlock()) {
+    auto *AI = dyn_cast<AllocaInst>(&I);
+    // Array allocations are probably not worth handling, since an allocation
+    // of the array type is the canonical form.
+    if (!AI || !AI->isStaticAlloca() || AI->isArrayAllocation())
+      continue;
+    LLVM_DEBUG(dbgs() << "Analyzing: " << *AI << '\n');
+    AllocaAnalysis AA{AI};
+    if (!collectAllocaUses(AA))
+      continue;
+    AllocaAnalysisMap.insert({AI, std::move(AA)});
+  }
 
-    for (int I = 0, E = Raw.size(); I != E; ++I) {
-      bool Promotable = true;
-      for (AllocaInst *Linked : Raw[I].Links) {
-        // If this alloca is not promotable, the whole group is not promotable
-        // as well.
-        if (Index[Linked] == -1) {
-          // The whole group can not be promoted
-          Promotable = false;
-          break;
-        }
-      }
+  std::vector<AllocaAnalysis> Allocas;
+  for (auto &AA : AllocaAnalysisMap.values()) {
+    if (AA.Uses.empty())
+      continue;
 
-      if (Promotable && !Raw[I].Uses.empty()) {
-        for (AllocaInst *Linked : Raw[I].Links) {
-          // We have an linked alloca which appears earlier, this is not leader.
-          //
-          assert(Index[Linked] > I && "This should be leader alloca\n");
-          // Pull from the other member alloca into the current alloca, which
-          // is the leader.
-          AllocaAnalysis &LeaderAA = Raw[I];
-          AllocaAnalysis &MemberAA = Raw[Index[Linked]];
-
-          // Merge the uses into the leader alloca.
-          append_range(LeaderAA.Uses, MemberAA.Uses);
-          // Clear member uses so that it will be skipped later.
-          MemberAA.Uses.clear();
-
-          LeaderAA.Pointers.insert_range(MemberAA.Pointers);
-          LeaderAA.Members.push_back(MemberAA.Alloca);
-          LeaderAA.HaveUnpromotableMerge |= MemberAA.HaveUnpromotableMerge;
-        }
-        // We only need to process leader alloca for later steps.
-        Allocas.push_back(Raw[I]);
+    bool Promotable = true;
+    SmallVector<AllocaInst *> Worklist;
+    LLVM_DEBUG(dbgs() << "Process leader alloca " << *AA.Alloca << '\n');
+    Worklist.append(AA.Links);
+    // Pull the information from links into the leader alloca.
+    while (!Worklist.empty()) {
+      auto *CurLink = Worklist.pop_back_val();
+      auto *LinkIter = AllocaAnalysisMap.find(CurLink);
+      if (LinkIter != AllocaAnalysisMap.end()) {
+        AllocaAnalysis &LinkAA = LinkIter->second;
+        // Skip if the linked alloca was done or points to the leader alloca
+        // itself.
+        if (LinkAA.Uses.empty() || LinkAA.Alloca == AA.Alloca)
+          continue;
+
+        LLVM_DEBUG({
+          dbgs() << "  Process link: " << *LinkAA.Alloca << '\n';
+          for (auto *X : LinkAA.Uses)
+            dbgs() << "    Add User " << *X->getUser() << '\n';
+        });
+
+        AA.Uses.insert_range(LinkAA.Uses);
+        LinkAA.Uses.clear();
+        AA.HaveUnpromotableMerge |= LinkAA.HaveUnpromotableMerge;
+        AA.Members.push_back(LinkAA.Alloca);
+
+        // Add indirect links to worklist, so that their info are properly
+        // propagated to the leader alloca.
+        Worklist.append(LinkAA.Links);
+        LinkAA.Links.clear();
+      } else {
+        LLVM_DEBUG(dbgs() << "  The alloca is not promotable\n");
+        Promotable = false;
+        break;
       }
     }
-  }
 
-  // Allocas that are merged by phi/select would have duplicated Use entries,
-  // remove them while preserving order.
-  for (AllocaAnalysis &AA : Allocas) {
-    if (!AA.isGroup())
-      continue;
-    SmallPtrSet<Use *, 16> Seen;
-    llvm::erase_if(AA.Uses, [&](Use *U) { return !Seen.insert(U).second; });
+    if (Promotable) {
+      sort(AA.Members, [](AllocaInst *A, AllocaInst *B) -> bool {
+        return A->comesBefore(B);
+      });
+      LLVM_DEBUG({
+        for (auto *M : AA.Members)
+          dbgs() << "  Members: " << *M << '\n';
+      });
+      Allocas.push_back(AA);
+    }
   }
 
   for (AllocaAnalysis &AA : Allocas) {
@@ -606,15 +557,11 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
   if (Cached != AA.Vector.IndexCache.end())
     return Cached->second;
 
-  const auto BaseLaneConst = [&](AllocaInst *Member) -> Value * {
-    return ConstantInt::get(Type::getInt32Ty(Ctx),
-                            AA.Vector.BaseLane.lookup(Member));
-  };
-
   // A pointer that is directly one of the member allocas indexes lane
   // BaseLane[member].
   if (auto *AI = dyn_cast<AllocaInst>(Ptr)) {
-    Value *Idx = BaseLaneConst(AI);
+    Value *Idx =
+        ConstantInt::get(Type::getInt32Ty(Ctx), AA.Vector.BaseLane.lookup(AI));
     AA.Vector.IndexCache[Ptr] = Idx;
     return Idx;
   }
@@ -623,8 +570,8 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
   // before recursing so self-referential phis terminate.
   if (auto *Phi = dyn_cast<PHINode>(Ptr)) {
     IRBuilder<> B(Phi);
-    PHINode *IdxPhi =
-        B.CreatePHI(B.getInt32Ty(), Phi->getNumIncomingValues(), "promotealloca.idx");
+    PHINode *IdxPhi = B.CreatePHI(B.getInt32Ty(), Phi->getNumIncomingValues(),
+                                  "promotealloca.idx");
     AA.Vector.IndexCache[Ptr] = IdxPhi;
     for (unsigned I = 0, E = Phi->getNumIncomingValues(); I != E; ++I)
       IdxPhi->addIncoming(calculateVectorIndex(Phi->getIncomingValue(I), AA),
@@ -691,6 +638,11 @@ static Value *calculateVectorIndex(Value *Ptr, AllocaAnalysis &AA) {
 static std::optional<GEPToVectorIndex>
 computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaAnalysis &AA,
                         Type *VecElemTy, const DataLayout &DL) {
+  auto &GEPVectorIndexMap = AA.Vector.GEPVectorIdx;
+  auto *IndexIter = GEPVectorIndexMap.find(GEP);
+  if (IndexIter != GEPVectorIndexMap.end())
+    return IndexIter->second;
+
   // TODO: Extracting a "multiple of X" from a GEP might be a useful generic
   // helper.
   LLVMContext &Ctx = GEP->getContext();
@@ -725,8 +677,7 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaAnalysis &AA,
   }
 
   // This should either points to the alloca or known phi/select.
-  assert(isa<AllocaInst>(CurPtr) ||
-         (isa<PHINode, SelectInst>(CurPtr) /*&& AA.Pointers.contains(CurPtr)*/));
+  assert(isa<AllocaInst>(CurPtr) || (isa<PHINode, SelectInst>(CurPtr)));
 
   int64_t VecElemSize = DL.getTypeAllocSize(VecElemTy);
   if (VarOffsets.size() > 1)
@@ -746,8 +697,10 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaAnalysis &AA,
     Result.ConstIndex = ConstantInt::get(Ctx, IndexQuot.sextOrTrunc(BW));
 
   // If there are no variable offsets, only a constant offset, then we're done.
-  if (VarOffsets.empty())
+  if (VarOffsets.empty()) {
+    GEPVectorIndexMap[GEP] = Result;
     return Result;
+  }
 
   // Scale is the stride in the (A * stride) part. Check that there is only one
   // variable offset and extract the scale factor.
@@ -792,6 +745,7 @@ computeGEPToVectorIndex(GetElementPtrInst *GEP, AllocaAnalysis &AA,
     Result.VarShift = ConstantInt::get(Ctx, APInt(BW, Log2_64(Divisor)));
   }
 
+  GEPVectorIndexMap[GEP] = Result;
   return Result;
 }
 
@@ -1214,8 +1168,8 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
 
     // Pointer phis/selects merging group members are handled by rewriting them
     // to lane-index phis/selects during promotion.
-    if (isa<PHINode, SelectInst>(Inst) && Inst->getType()->isPointerTy()) {
-      AA.Vector.PtrMerges.insert(Inst);
+    if (isa<PHINode, SelectInst>(Inst)) {
+      AA.Vector.UsersToRemove.insert(Inst);
       continue;
     }
 
@@ -1258,8 +1212,7 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
       if (!Index)
         return RejectUser(Inst, "cannot compute vector index for GEP");
 
-      AA.Vector.GEPVectorIdx[GEP] = std::move(Index.value());
-      AA.Vector.UsersToRemove.push_back(Inst);
+      AA.Vector.UsersToRemove.insert(Inst);
       continue;
     }
 
@@ -1284,12 +1237,13 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
         if (Ptr == AA.Alloca)
           return ConstantInt::get(Ptr->getContext(), APInt(32, 0));
 
-        GetElementPtrInst *GEP = cast<GetElementPtrInst>(Ptr);
-        const auto &GEPI = AA.Vector.GEPVectorIdx.find(GEP)->second;
-        if (GEPI.VarIndex)
+        auto Index = computeGEPToVectorIndex(cast<GetElementPtrInst>(Ptr), AA,
+                                             VecEltTy, DL);
+
+        if (!Index || Index->VarIndex || Index->BasePtr != AA.Alloca)
           return nullptr;
-        if (GEPI.ConstIndex)
-          return GEPI.ConstIndex;
+        if (Index->ConstIndex)
+          return Index->ConstIndex;
         return ConstantInt::get(Ptr->getContext(), APInt(32, 0));
       };
 
@@ -1324,14 +1278,14 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
     if (isAssumeLikeIntrinsic(Inst)) {
       if (!Inst->use_empty())
         return RejectUser(Inst, "assume-like intrinsic cannot have any users");
-      AA.Vector.UsersToRemove.push_back(Inst);
+      AA.Vector.UsersToRemove.insert(Inst);
       continue;
     }
 
     if (isa<ICmpInst>(Inst) && all_of(Inst->users(), [](User *U) {
           return isAssumeLikeIntrinsic(cast<Instruction>(U));
         })) {
-      AA.Vector.UsersToRemove.push_back(Inst);
+      AA.Vector.UsersToRemove.insert(Inst);
       continue;
     }
 
@@ -1420,20 +1374,10 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
     I->eraseFromParent();
   }
 
-  // The pointer phis/selects merging group members are now dead. Drop them
-  // before the GEPs they reference. RAUW with poison first so chained merges
-  // can be erased in any order.
-  for (Instruction *I : AA.Vector.PtrMerges)
-    I->replaceAllUsesWith(PoisonValue::get(I->getType()));
-  for (Instruction *I : AA.Vector.PtrMerges) {
-    assert(I->use_empty());
-    I->eraseFromParent();
-  }
-
   // Delete all the users that are known to be removeable.
-  for (Instruction *I : reverse(AA.Vector.UsersToRemove)) {
-    I->dropDroppableUses();
-    assert(I->use_empty());
+  // Replace the uses with poison first so they can be deleted in any order.
+  for (Instruction *I : AA.Vector.UsersToRemove) {
+    I->replaceAllUsesWith(PoisonValue::get(I->getType()));
     I->eraseFromParent();
   }
 
@@ -1985,8 +1929,8 @@ bool AMDGPUPromoteAllocaImpl::tryPromoteAllocaToLDS(
     case Intrinsic::invariant_end:
     case Intrinsic::launder_invariant_group:
     case Intrinsic::strip_invariant_group: {
-      assert(Intr->getArgOperand(Intr->arg_size() - 1)->getType() == NewPtrTy &&
-             "pointer operand should already have been promoted");
+      // Since the worklist can be in any order, the argument may still be the
+      // old type. Its type will be fixed when processing its definition.
       Function *NewF = Intrinsic::getOrInsertDeclaration(
           Intr->getModule(), Intr->getIntrinsicID(), NewPtrTy);
       Intr->mutateType(NewF->getReturnType());
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll
index 473fa92547a43..922287a10b73c 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-scoring.ll
@@ -1,16 +1,14 @@
 ; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -debug-only=amdgpu-promote-alloca -amdgpu-promote-alloca-to-vector-limit=512 -passes=amdgpu-promote-alloca %s -o - 2>&1 | FileCheck %s
 ; REQUIRES: asserts
 
-; CHECK-LABEL: Analyzing:   %simpleuser = alloca [4 x i64], align 4, addrspace(5)
-; CHECK-NEXT: Analyzing:   %manyusers = alloca [4 x i64], align 4, addrspace(5)
-; CHECK-NEXT: Scoring:   %simpleuser = alloca [4 x i64], align 4, addrspace(5)
+; CHECK-LABEL: Scoring:   %simpleuser = alloca [4 x i64], align 4, addrspace(5)
 ; CHECK-NEXT:   [+1]:   store i64 42, ptr addrspace(5) %simpleuser, align 8
 ; CHECK-NEXT:   => Final Score:1
 ; CHECK-NEXT: Scoring:   %manyusers = alloca [4 x i64], align 4, addrspace(5)
-; CHECK-NEXT:   [+1]:   store i64 %v0.add, ptr addrspace(5) %manyusers.1, align 8
-; CHECK-NEXT:   [+1]:   %v0 = load i64, ptr addrspace(5) %manyusers.1, align 8
-; CHECK-NEXT:   [+1]:   store i64 %v1.add, ptr addrspace(5) %manyusers.2, align 8
-; CHECK-NEXT:   [+1]:   %v1 = load i64, ptr addrspace(5) %manyusers.2, align 8
+; CHECK-DAG:   [+1]:   store i64 %v0.add, ptr addrspace(5) %manyusers.1, align 8
+; CHECK-DAG:   [+1]:   %v0 = load i64, ptr addrspace(5) %manyusers.1, align 8
+; CHECK-DAG:   [+1]:   store i64 %v1.add, ptr addrspace(5) %manyusers.2, align 8
+; CHECK-DAG:   [+1]:   %v1 = load i64, ptr addrspace(5) %manyusers.2, align 8
 ; CHECK-NEXT:   => Final Score:4
 ; CHECK-NEXT: Sorted Worklist:
 ; CHECK-NEXT:     %manyusers = alloca [4 x i64], align 4, addrspace(5)
@@ -37,14 +35,13 @@ entry:
   ret void
 }
 
-; CHECK-LABEL: Analyzing:   %stack = alloca [4 x i64], align 4, addrspace(5)
-; CHECK-NEXT: Scoring:   %stack = alloca [4 x i64], align 4, addrspace(5)
-; CHECK-NEXT:   [+5]:   store i64 32, ptr addrspace(5) %stack, align 8
-; CHECK-NEXT:   [+1]:   store i64 42, ptr addrspace(5) %stack, align 8
-; CHECK-NEXT:   [+9]:   store i64 32, ptr addrspace(5) %stack.1, align 8
-; CHECK-NEXT:   [+5]:   %outer = load i64, ptr addrspace(5) %stack.1, align 8
-; CHECK-NEXT:   [+1]:   store i64 64, ptr addrspace(5) %stack.2, align 8
-; CHECK-NEXT:   [+9]:   %inner = load i64, ptr addrspace(5) %stack.2, align 8
+; CHECK-LABEL: Scoring:   %stack = alloca [4 x i64], align 4, addrspace(5)
+; CHECK-DAG:   [+5]:   store i64 32, ptr addrspace(5) %stack, align 8
+; CHECK-DAG:   [+1]:   store i64 42, ptr addrspace(5) %stack, align 8
+; CHECK-DAG:   [+9]:   store i64 32, ptr addrspace(5) %stack.1, align 8
+; CHECK-DAG:   [+5]:   %outer = load i64, ptr addrspace(5) %stack.1, align 8
+; CHECK-DAG:   [+1]:   store i64 64, ptr addrspace(5) %stack.2, align 8
+; CHECK-DAG:   [+9]:   %inner = load i64, ptr addrspace(5) %stack.2, align 8
 ; CHECK-NEXT:   => Final Score:30
 define amdgpu_kernel void @loop_users_alloca(i1 %x, i2) #0 {
 entry:
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
index b57874848e497..ab79cc2fd3119 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-phi.ll
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -S -mtriple=amdgpu-unknown-amdhsa -passes=amdgpu-promote-alloca-to-vector < %s | FileCheck %s
+; RUN: opt -S -mtriple=amdgpu7.00-unknown-amdhsa -passes=amdgpu-promote-alloca-to-vector < %s | FileCheck %s
 
 ; Tests for promotion of allocas whose pointers are merged by phis/selects,
 
@@ -38,6 +38,50 @@ join:
   ret void
 }
 
+define amdgpu_kernel void @phi_three_allocas(i1 %f, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_three_allocas(
+; CHECK-SAME: i1 [[F:%.*]], ptr addrspace(1) [[OUT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[A:%.*]] = freeze <3 x i32> poison
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <3 x i32> [[A]], i32 1, i32 0
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <3 x i32> [[TMP0]], i32 2, i32 1
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <3 x i32> [[TMP1]], i32 3, i32 2
+; CHECK-NEXT:    br i1 [[F]], label %[[BB_1:.*]], label %[[BB_2:.*]]
+; CHECK:       [[BB_1]]:
+; CHECK-NEXT:    br label %[[JOIN:.*]]
+; CHECK:       [[BB_2]]:
+; CHECK-NEXT:    br label %[[JOIN]]
+; CHECK:       [[JOIN]]:
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX:%.*]] = phi i32 [ 0, %[[BB_1]] ], [ 1, %[[BB_2]] ]
+; CHECK-NEXT:    [[PROMOTEALLOCA_IDX1:%.*]] = phi i32 [ 1, %[[BB_1]] ], [ 2, %[[BB_2]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <3 x i32> [[TMP2]], i32 [[PROMOTEALLOCA_IDX]]
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <3 x i32> [[TMP2]], i32 [[PROMOTEALLOCA_IDX1]]
+; CHECK-NEXT:    [[L:%.*]] = add i32 [[TMP3]], [[TMP4]]
+; CHECK-NEXT:    store i32 [[L]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %a = alloca i32, addrspace(5)
+  %b = alloca i32, addrspace(5)
+  %c = alloca i32, addrspace(5)
+  store i32 1, ptr addrspace(5) %a
+  store i32 2, ptr addrspace(5) %b
+  store i32 3, ptr addrspace(5) %c
+  br i1 %f, label %bb.1, label %bb.2
+bb.1:
+  br label %join
+bb.2:
+  br label %join
+join:
+  %p = phi ptr addrspace(5) [ %a, %bb.1 ], [ %b, %bb.2 ]
+  %q = phi ptr addrspace(5) [ %b, %bb.1 ], [ %c, %bb.2 ]
+  %l1 = load i32, ptr addrspace(5) %p
+  %l2 = load i32, ptr addrspace(5) %q
+  %l = add i32 %l1, %l2
+  store i32 %l, ptr addrspace(1) %out
+  ret void
+}
+
 ; GEP uses a (chained) pointer phi: phi-of-phi feeding a gep.
 define amdgpu_kernel void @gep_uses_chained_phi(i1 %c, i1 %d, i32 %v, ptr addrspace(1) %out) {
 ; CHECK-LABEL: define amdgpu_kernel void @gep_uses_chained_phi(
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll
index 8b88768f6ae7a..74a7398be0558 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll
@@ -1,4 +1,4 @@
-; RUN: llc -mtriple=amdgcn-amd-amdhsa < %s | FileCheck %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -disable-promote-alloca-to-vector < %s | FileCheck %s
 
 define i64 @i64_test(i64 %i) nounwind readnone {
   %loc = alloca i64, addrspace(5)
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll.expected b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll.expected
index 53a164ed71759..6be3b24a735f3 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_asm.ll.expected
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -mtriple=amdgcn-amd-amdhsa < %s | FileCheck %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -disable-promote-alloca-to-vector < %s | FileCheck %s
 
 define i64 @i64_test(i64 %i) nounwind readnone {
 ; CHECK-LABEL: i64_test:
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll
index 6381aaf63c84e..42d2eb3984f10 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll
@@ -1,4 +1,4 @@
-; RUN: llc -enable-machine-outliner -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
+; RUN: llc -enable-machine-outliner -disable-promote-alloca-to-vector -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
 
 ; NOTE: Machine outliner doesn't run.
 @x = dso_local global i32 0, align 4
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.generated.expected b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.generated.expected
index 0a85133152679..02d207f5c8609 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.generated.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.generated.expected
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --include-generated-funcs
-; RUN: llc -enable-machine-outliner -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
+; RUN: llc -enable-machine-outliner -disable-promote-alloca-to-vector -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
 
 ; NOTE: Machine outliner doesn't run.
 @x = dso_local global i32 0, align 4
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.nogenerated.expected b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.nogenerated.expected
index df156b1b2e1b4..c913aeb16fbf8 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.nogenerated.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_generated_funcs.ll.nogenerated.expected
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -enable-machine-outliner -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
+; RUN: llc -enable-machine-outliner -disable-promote-alloca-to-vector -mtriple=amdgcn-adm-amdhsa < %s | FileCheck %s
 
 ; NOTE: Machine outliner doesn't run.
 @x = dso_local global i32 0, align 4
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll
index eabcbcbc79528..8f3e3e1f5eda9 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll
@@ -1,4 +1,4 @@
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -stop-after=finalize-isel -debug-only=isel -o /dev/null %s 2>&1 | FileCheck %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -disable-promote-alloca-to-vector -stop-after=finalize-isel -debug-only=isel -o /dev/null %s 2>&1 | FileCheck %s
 
 define i64 @i64_test(i64 %i) nounwind readnone {
   %loc = alloca i64, addrspace(5)
diff --git a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll.expected b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll.expected
index ba42798c5c9be..ada56f9b08d12 100644
--- a/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_llc_test_checks/Inputs/amdgpu_isel.ll.expected
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -stop-after=finalize-isel -debug-only=isel -o /dev/null %s 2>&1 | FileCheck %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -disable-promote-alloca-to-vector -stop-after=finalize-isel -debug-only=isel -o /dev/null %s 2>&1 | FileCheck %s
 
 define i64 @i64_test(i64 %i) nounwind readnone {
 ; CHECK-LABEL: i64_test:

>From 0d6770bb9d2d09f3351652ddc73fc4f147fc3b24 Mon Sep 17 00:00:00 2001
From: Ruiling Song <ruiling.song at amd.com>
Date: Fri, 11 Sep 2026 15:23:08 +0800
Subject: [PATCH 3/3] Fix build failures

---
 llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp             | 2 +-
 .../AMDGPU/promote-alloca-proper-value-replacement.ll      | 7 +++----
 2 files changed, 4 insertions(+), 5 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index 48a793b0174a2..dcad3bdb5df4f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -1149,7 +1149,6 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
       ElemCnt = cast<FixedVectorType>(MemberTy)->getNumElements();
     }
 
-
     AA.Vector.BaseLane[Member] = TotalElems;
     TotalElems += ElemCnt;
   }
@@ -1388,6 +1387,7 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
   // Delete all the users that are known to be removeable.
   // Replace the uses with poison first so they can be deleted in any order.
   for (Instruction *I : AA.Vector.UsersToRemove) {
+    I->dropDroppableUses();
     I->replaceAllUsesWith(PoisonValue::get(I->getType()));
     I->eraseFromParent();
   }
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-proper-value-replacement.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-proper-value-replacement.ll
index 8668c4721faa8..befd06ea429b3 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-proper-value-replacement.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-proper-value-replacement.ll
@@ -7,8 +7,8 @@ define void @alloca_value_cross_reference() {
 ; CHECK-NEXT:    [[HIT_ORDERED:%.*]] = freeze <4 x float> poison
 ; CHECK-NEXT:    [[HIT_INDEX:%.*]] = freeze <4 x i32> poison
 ; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> [[HIT_INDEX]], i32 1, i32 0
-; CHECK-NEXT:    br [[DOTLR_PH5:label %.*]]
-; CHECK:       [[_LR_PH5:.*:]]
+; CHECK-NEXT:    br label %[[DOTLR_PH5:.*]]
+; CHECK:       [[DOTLR_PH5]]:
 ; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x i32> [[TMP0]], i32 0
 ; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x float> [[HIT_ORDERED]], float 0.000000e+00, i32 [[TMP1]]
 ; CHECK-NEXT:    ret void
@@ -41,10 +41,9 @@ define half @forwarded_load_across_blocks() {
 ; CHECK-NEXT:    [[ARR:%.*]] = freeze <4 x half> poison
 ; CHECK-NEXT:    br label %[[BB2:.*]]
 ; CHECK:       [[BB2]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = freeze <4 x half> <half 1.000000e+00, half 2.000000e+00, half 3.000000e+00, half 4.000000e+00>
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x half> [[TMP0]], i32 0
+; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x half> <half 1.000000e+00, half 2.000000e+00, half 3.000000e+00, half 4.000000e+00>, i32 0
 ; CHECK-NEXT:    ret half [[TMP1]]
 ;
 entry:



More information about the llvm-commits mailing list