[llvm-branch-commits] [llvm] [AMDGPU] Promote private objects into the VGPR address space via flag (PR #215282)

Gheorghe-Teodor Bercea via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Tue Aug 11 11:00:14 PDT 2026


https://github.com/doru1004 updated https://github.com/llvm/llvm-project/pull/215282

>From 30311818aa89b8126e6d0bb0ca4b971f024ec0a0 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Mon, 10 Aug 2026 15:45:03 +0200
Subject: [PATCH 1/2] Promote private objects into the VGPR address space via
 flag

---
 .../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp |  29 +-
 llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.cpp  |  17 ++
 llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.h    |  19 +-
 .../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 141 ++++++++-
 .../AddressSpaceVGPR/as-vgpr-promote.ll       | 279 ++++++++++++++++++
 5 files changed, 453 insertions(+), 32 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-promote.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index 5ba8c02f00b20..0f36d874d7c69 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -556,33 +556,6 @@ static bool isLoadStoreLegal(const GCNSubtarget &ST, const LegalityQuery &Query)
          !hasBufferRsrcWorkaround(Ty) && !loadStoreBitcastWorkaround(Ty);
 }
 
-// Whether the VGPR ("as memory") load/store lowering handles a MemSize-bit
-// memory access producing/consuming a ValSize-bit value at the given alignment.
-// Whole-dword accesses (those with a matching V_LOAD_IDX/V_STORE_IDX pseudo)
-// are supported, as are 8-/16-bit accesses, including extending loads into a
-// 16- or 32-bit value.
-//
-// A sub-dword access is implemented as a bit-field extract from (or insert
-// into) the dword containing it, so it must not straddle a dword boundary. An
-// 8-bit access never can; a 16-bit one only if it is 2-byte aligned. Requiring
-// natural alignment covers both, and is what lets the bit offset within the
-// dword be computed from a possibly dynamic pointer.
-static bool isVGPRLoadStoreSupported(unsigned MemSize, unsigned ValSize,
-                                     Align Alignment) {
-  if (MemSize == 8 || MemSize == 16) {
-    if (Alignment < Align(MemSize / 8))
-      return false;
-    if (ValSize == MemSize)
-      return true;
-    if (ValSize > MemSize && (ValSize == 16 || ValSize == 32))
-      return true;
-    return false;
-  }
-  if (MemSize != ValSize)
-    return false;
-  return AMDGPUMI::VLoadIdxInst::tryGetOpcodeForBitWidth(MemSize) != -1;
-}
-
 /// Return true if a load or store of the type should be lowered with a bitcast
 /// to a different type.
 static bool shouldBitcastLoadStoreType(const GCNSubtarget &ST, const LLT Ty,
@@ -3564,7 +3537,7 @@ static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
 
   // Whole-dword and naturally aligned 8-/16-bit accesses are implemented.
   // Reject anything else with a diagnostic instead of failing to legalize.
-  if (!isVGPRLoadStoreSupported(MemSize, ValSize, MMO.getAlign())) {
+  if (!AMDGPU::isVGPRLoadStoreSupported(MemSize, ValSize, MMO.getAlign())) {
     const Function &F = B.getMF().getFunction();
     F.getContext().diagnose(DiagnosticInfoUnsupported(
         F,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.cpp
index 7614fc23fac40..507229c67825a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.cpp
@@ -8,6 +8,7 @@
 
 #include "AMDGPUMemoryUtils.h"
 #include "AMDGPU.h"
+#include "AMDGPUMachineInstrs.h"
 #include "Utils/AMDGPUBaseInfo.h"
 #include "llvm/ADT/SetOperations.h"
 #include "llvm/Analysis/AliasAnalysis.h"
@@ -46,6 +47,22 @@ unsigned AllocatedVGPRsMetadata::getSize() const {
       ->getZExtValue();
 }
 
+bool isVGPRLoadStoreSupported(unsigned MemSize, unsigned ValSize,
+                              Align Alignment) {
+  if (MemSize == 8 || MemSize == 16) {
+    if (Alignment < Align(MemSize / 8))
+      return false;
+    if (ValSize == MemSize)
+      return true;
+    if (ValSize > MemSize && (ValSize == 16 || ValSize == 32))
+      return true;
+    return false;
+  }
+  if (MemSize != ValSize)
+    return false;
+  return AMDGPUMI::VLoadIdxInst::tryGetOpcodeForBitWidth(MemSize) != -1;
+}
+
 bool AllocatedVGPRsMetadata::classof(const MDNode *N) {
   if (N->getNumOperands() != 2)
     return false;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.h b/llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.h
index db04c254feaeb..dea2257df2505 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.h
@@ -13,10 +13,10 @@
 #include "llvm/ADT/DenseMap.h"
 #include "llvm/ADT/DenseSet.h"
 #include "llvm/IR/Metadata.h"
+#include "llvm/Support/Alignment.h"
 
 namespace llvm {
 
-struct Align;
 class AAResults;
 class AllocaInst;
 class DataLayout;
@@ -125,6 +125,23 @@ class AllocatedVGPRsMetadata : public MDNode {
   static bool classof(const MDNode *N);
 };
 
+/// Whether the VGPR ("as memory") address space implements a \p MemSize-bit
+/// memory access producing/consuming a \p ValSize-bit value at the given
+/// alignment. Whole-dword accesses (those with a matching V_LOAD_IDX /
+/// V_STORE_IDX pseudo) are supported, as are 8-/16-bit accesses, including
+/// extending loads into a 16- or 32-bit value.
+///
+/// A sub-dword access is implemented as a bit-field extract from (or insert
+/// into) the dword containing it, so it must not straddle a dword boundary. An
+/// 8-bit access never can; a 16-bit one only if it is 2-byte aligned. Requiring
+/// natural alignment covers both, and is what lets the bit offset within the
+/// dword be computed from a possibly dynamic pointer.
+///
+/// Lowering diagnoses an access this rejects, so anything deciding to put an
+/// object in this address space has to agree with it.
+bool isVGPRLoadStoreSupported(unsigned MemSize, unsigned ValSize,
+                              Align Alignment);
+
 } // end namespace AMDGPU
 
 } // end namespace llvm
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index dc64a33187296..27be32ee05d40 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -37,6 +37,7 @@
 #include "llvm/Analysis/ValueTracking.h"
 #include "llvm/CodeGen/TargetPassConfig.h"
 #include "llvm/IR/IRBuilder.h"
+#include "llvm/IR/InstIterator.h"
 #include "llvm/IR/IntrinsicInst.h"
 #include "llvm/IR/IntrinsicsAMDGPU.h"
 #include "llvm/IR/IntrinsicsR600.h"
@@ -87,6 +88,14 @@ static cl::opt<unsigned>
                             "when sorting profitable allocas"),
                    cl::init(4));
 
+// An object in the VGPR ("as memory") address space cannot be spilled, so one
+// that is live across a call has nowhere to go: see analyzePromoteToVGPR.
+// TODO: Enable by default once objects can survive a call.
+static cl::opt<bool>
+    EnablePromoteToVGPR("amdgpu-promote-private",
+                        cl::desc("Enable promoting private objects into VGPRs"),
+                        cl::init(false), cl::Hidden);
+
 // We support vector indices of the form ((A * stride) >> shift) + B
 // VarIndex is A, VarMul is stride, VarShift is shift and ConstIndex is B. All
 // parts are optional.
@@ -121,6 +130,9 @@ struct AllocaAnalysis {
     bool Enable = false;
     SmallVector<User *> Worklist;
   } LDS;
+  struct {
+    bool Enable = false;
+  } VGPR;
 
   explicit AllocaAnalysis(AllocaInst *Alloca) : Alloca(Alloca) {}
 };
@@ -165,6 +177,8 @@ class AMDGPUPromoteAllocaImpl {
   FixedVectorType *getVectorTypeForAlloca(Type *AllocaTy) const;
   void analyzePromoteToVector(AllocaAnalysis &AA) const;
   void promoteAllocaToVector(AllocaAnalysis &AA);
+  void analyzePromoteToVGPR(AllocaAnalysis &AA) const;
+  void promoteAllocaToVGPR(AllocaAnalysis &AA);
   void analyzePromoteToLDS(AllocaAnalysis &AA) const;
   bool tryPromoteAllocaToLDS(AllocaAnalysis &AA, bool SufficientLDS,
                              SetVector<IntrinsicInst *> &DeferredIntrs);
@@ -405,6 +419,27 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool IsLatePass, bool NoOpt) {
     return false;
 
   const bool PromoteToLDS = IsLatePass && !NoOpt;
+  bool PromoteToVGPR = EnablePromoteToVGPR && IsLatePass && !NoOpt;
+
+  // An object in the VGPR ("as memory") address space occupies fixed registers
+  // for the whole of its live range. Those registers are caller-saved and the
+  // object cannot be spilled, so one that is live across a call has nowhere to
+  // be, and AMDGPUPrivateObjectVGPRs diagnoses it. Promoting nothing in a
+  // function that makes a call keeps this from turning working code into an
+  // error.
+  //
+  // Intrinsics do not count: on this target they lower to instructions rather
+  // than to calls. Were one ever to lower to a call, the result would be that
+  // diagnostic rather than a wrong answer.
+  //
+  // TODO: This is conservative. Only a call the object is live across matters,
+  // and then only one that does not preserve the registers it occupies.
+  if (PromoteToVGPR) {
+    PromoteToVGPR = none_of(instructions(F), [](const Instruction &I) {
+      const auto *CB = dyn_cast<CallBase>(&I);
+      return CB && !isa<IntrinsicInst>(CB);
+    });
+  }
 
   bool SufficientLDS = PromoteToLDS && hasSufficientLocalMem(F);
   MaxVGPRs = IsAMDGCN ? getMaxVGPRs(CurrentLocalMemUsage, TM, F) : 128;
@@ -443,9 +478,11 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool IsLatePass, bool NoOpt) {
 
       if (collectAllocaUses(AA)) {
         analyzePromoteToVector(AA);
+        if (PromoteToVGPR)
+          analyzePromoteToVGPR(AA);
         if (PromoteToLDS)
           analyzePromoteToLDS(AA);
-        if (AA.Vector.Ty || AA.LDS.Enable) {
+        if (AA.Vector.Ty || AA.LDS.Enable || AA.VGPR.Enable) {
           scoreAlloca(AA);
           Allocas.push_back(std::move(AA));
         }
@@ -485,13 +522,19 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool IsLatePass, bool NoOpt) {
       continue;
     }
 
-    if (AA.Vector.Ty) {
+    // Vectorization and promotion into the VGPR address space both spend the
+    // same registers, so they draw on the same budget. Vectorization is
+    // preferred where an alloca qualifies for either.
+    if (AA.Vector.Ty || AA.VGPR.Enable) {
       std::optional<TypeSize> Size = AA.Alloca->getAllocationSize(DL);
       assert(Size); // Expected to succeed on non-array alloca.
       const unsigned AllocaCost = Size->getFixedValue() * 8;
       // First, check if we have enough budget to vectorize this alloca.
       if (AllocaCost <= VectorizationBudget) {
-        promoteAllocaToVector(AA);
+        if (AA.Vector.Ty)
+          promoteAllocaToVector(AA);
+        else
+          promoteAllocaToVGPR(AA);
         Changed = true;
         assert((VectorizationBudget - AllocaCost) < VectorizationBudget &&
                "Underflow!");
@@ -1321,6 +1364,98 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
   AA.Alloca->eraseFromParent();
 }
 
+// Decide whether an alloca can be moved into the VGPR ("as memory") address
+// space, where it lives in registers rather than in scratch and is reached with
+// an indexed register access instead of a load or store.
+void AMDGPUPromoteAllocaImpl::analyzePromoteToVGPR(AllocaAnalysis &AA) const {
+  if (!IsAMDGCN)
+    return;
+
+  const auto Reject = [&](const Instruction *Inst, Twine Msg) {
+    LLVM_DEBUG(dbgs() << "  Cannot promote alloca to VGPRs: " << Msg << "\n"
+                      << "    " << *Inst << "\n");
+  };
+
+  for (Use *U : AA.Uses) {
+    Instruction *Inst = cast<Instruction>(U->getUser());
+
+    if (getLoadStorePointerOperand(Inst)) {
+      assert(!isa<StoreInst>(Inst) ||
+             U->getOperandNo() == StoreInst::getPointerOperandIndex());
+
+      bool IsSimple = isa<LoadInst>(Inst) ? cast<LoadInst>(Inst)->isSimple()
+                                          : cast<StoreInst>(Inst)->isSimple();
+      if (!IsSimple)
+        return Reject(Inst, "not a simple load or store");
+
+      // Promoting an access the lowering cannot implement would turn this into
+      // a diagnostic, so ask the lowering rather than guessing.
+      TypeSize AccessSize = DL.getTypeSizeInBits(getLoadStoreType(Inst));
+      if (AccessSize.isScalable())
+        return Reject(Inst, "scalable access");
+
+      // The value and memory sizes are the same here: an extending load of an
+      // object in private memory is a plain load followed by an extend.
+      unsigned Bits = AccessSize.getFixedValue();
+      Align Alignment = isa<LoadInst>(Inst) ? cast<LoadInst>(Inst)->getAlign()
+                                            : cast<StoreInst>(Inst)->getAlign();
+      if (!AMDGPU::isVGPRLoadStoreSupported(Bits, Bits, Alignment))
+        return Reject(Inst, "unsupported access size or alignment");
+
+      continue;
+    }
+
+    // These only compute addresses; collectAllocaUses has already established
+    // that a select or phi does not mix objects.
+    if (isa<GetElementPtrInst, SelectInst, PHINode>(Inst))
+      continue;
+
+    if (auto *MSI = dyn_cast<MemSetInst>(Inst)) {
+      if (!isSupportedMemset(MSI, AA.Alloca, DL))
+        return Reject(MSI, "cannot handle partial memset");
+      continue;
+    }
+
+    if (isa<MemTransferInst>(Inst))
+      return Reject(Inst, "cannot handle mem transfer");
+
+    if (auto *Intr = dyn_cast<IntrinsicInst>(Inst)) {
+      if (Intr->getIntrinsicID() == Intrinsic::objectsize)
+        continue;
+
+      if (isAssumeLikeIntrinsic(Inst)) {
+        if (!Inst->use_empty())
+          return Reject(Inst, "assume-like intrinsic cannot have any users");
+        continue;
+      }
+    }
+
+    // A comparison whose only purpose is to feed an assume.
+    if (isa<ICmpInst>(Inst) && all_of(Inst->users(), [](User *U) {
+          return isAssumeLikeIntrinsic(cast<Instruction>(U));
+        }))
+      continue;
+
+    return Reject(Inst, "unhandled alloca user");
+  }
+
+  AA.VGPR.Enable = true;
+}
+
+// Move the alloca into the VGPR ("as memory") address space. Pointers into it
+// are the same size in both address spaces, so the pointers derived from it
+// only need their type changed, and allocateVgprs then gives the object its
+// place in that address space.
+void AMDGPUPromoteAllocaImpl::promoteAllocaToVGPR(AllocaAnalysis &AA) {
+  LLVM_DEBUG(dbgs() << "Promoting alloca to VGPRs: " << *AA.Alloca << '\n');
+
+  Type *PtrTy = PointerType::get(Mod.getContext(), AMDGPUAS::VGPR);
+  for (Value *Ptr : AA.Pointers)
+    Ptr->mutateType(PtrTy);
+
+  allocateVgprs(AA);
+}
+
 std::pair<Value *, Value *>
 AMDGPUPromoteAllocaImpl::getLocalSizeYZ(IRBuilder<> &Builder) {
   Function &F = *Builder.GetInsertBlock()->getParent();
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-promote.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-promote.ll
new file mode 100644
index 0000000000000..c0b15b5f954e7
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-promote.ll
@@ -0,0 +1,279 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu12.00-- -passes=amdgpu-promote-alloca \
+; RUN:     -amdgpu-promote-private -o - %s | FileCheck %s
+; RUN: opt -S -mtriple=amdgpu9.42-- -passes=amdgpu-promote-alloca \
+; RUN:     -amdgpu-promote-private -o - %s | FileCheck %s
+; RUN: opt -S -mtriple=amdgpu12.00-- -passes=amdgpu-promote-alloca -o - %s \
+; RUN:     | FileCheck %s --check-prefix=OFF
+
+; A private alloca can be moved into the VGPR "as memory" address space (13),
+; where it lives in registers rather than in scratch. The pointers derived from
+; it change address space and the object is then allocated exactly as one
+; written in that address space to begin with.
+;
+; Promotion is behind -amdgpu-promote-private, so the OFF prefix checks that
+; nothing moves without it. It needs no target feature beyond what the address
+; space itself needs, so the two triples cover both the movrel and the
+; VGPR-index-mode lowerings.
+
+; The object is indexed at byte granularity, which vectorization cannot express,
+; so this reaches the VGPR path rather than being turned into a vector.
+define amdgpu_kernel void @promoted(ptr addrspace(1) %out, i32 %i) {
+; CHECK-LABEL: define amdgpu_kernel void @promoted(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; CHECK-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(13), !amdgpu.allocated.vgprs [[META0:![0-9]+]]
+; CHECK-NEXT:    call void @llvm.amdgcn.vgpr.lifetime.start.p13(ptr addrspace(13) [[OBJ]])
+; CHECK-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(13) [[OBJ]], i32 [[I]]
+; CHECK-NEXT:    store i32 7, ptr addrspace(13) [[P]], align 4
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr addrspace(13) [[P]], align 4
+; CHECK-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+; OFF-LABEL: define amdgpu_kernel void @promoted(
+; OFF-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; OFF-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; OFF-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; OFF-NEXT:    store i32 7, ptr addrspace(5) [[P]], align 4
+; OFF-NEXT:    [[V:%.*]] = load i32, ptr addrspace(5) [[P]], align 4
+; OFF-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; OFF-NEXT:    ret void
+;
+  %obj = alloca [8 x i32], align 4, addrspace(5)
+  %p = getelementptr i8, ptr addrspace(5) %obj, i32 %i
+  store i32 7, ptr addrspace(5) %p, align 4
+  %v = load i32, ptr addrspace(5) %p, align 4
+  store i32 %v, ptr addrspace(1) %out
+  ret void
+}
+
+; Sub-dword accesses are supported, so a mix of widths does not prevent this.
+define amdgpu_kernel void @promoted_sub_dword(ptr addrspace(1) %out, i32 %i) {
+; CHECK-LABEL: define amdgpu_kernel void @promoted_sub_dword(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; CHECK-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(13), !amdgpu.allocated.vgprs [[META0]]
+; CHECK-NEXT:    call void @llvm.amdgcn.vgpr.lifetime.start.p13(ptr addrspace(13) [[OBJ]])
+; CHECK-NEXT:    [[P:%.*]] = getelementptr [8 x i32], ptr addrspace(13) [[OBJ]], i32 0, i32 [[I]]
+; CHECK-NEXT:    store i32 7, ptr addrspace(13) [[P]], align 4
+; CHECK-NEXT:    [[Q:%.*]] = getelementptr i8, ptr addrspace(13) [[OBJ]], i32 [[I]]
+; CHECK-NEXT:    [[V:%.*]] = load i8, ptr addrspace(13) [[Q]], align 1
+; CHECK-NEXT:    [[Z:%.*]] = zext i8 [[V]] to i32
+; CHECK-NEXT:    store i32 [[Z]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+; OFF-LABEL: define amdgpu_kernel void @promoted_sub_dword(
+; OFF-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; OFF-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; OFF-NEXT:    [[P:%.*]] = getelementptr [8 x i32], ptr addrspace(5) [[OBJ]], i32 0, i32 [[I]]
+; OFF-NEXT:    store i32 7, ptr addrspace(5) [[P]], align 4
+; OFF-NEXT:    [[Q:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; OFF-NEXT:    [[V:%.*]] = load i8, ptr addrspace(5) [[Q]], align 1
+; OFF-NEXT:    [[Z:%.*]] = zext i8 [[V]] to i32
+; OFF-NEXT:    store i32 [[Z]], ptr addrspace(1) [[OUT]], align 4
+; OFF-NEXT:    ret void
+;
+  %obj = alloca [8 x i32], align 4, addrspace(5)
+  %p = getelementptr [8 x i32], ptr addrspace(5) %obj, i32 0, i32 %i
+  store i32 7, ptr addrspace(5) %p
+  %q = getelementptr i8, ptr addrspace(5) %obj, i32 %i
+  %v = load i8, ptr addrspace(5) %q
+  %z = zext i8 %v to i32
+  store i32 %z, ptr addrspace(1) %out
+  ret void
+}
+
+declare void @extern()
+
+; An object in this address space cannot be spilled and the registers it
+; occupies are caller-saved, so one live across a call has nowhere to be and the
+; backend rejects it. Promotion therefore declines rather than turning working
+; code into an error. This is conservative: the call need not be in the object's
+; live range for promotion to be refused.
+define amdgpu_kernel void @not_promoted_call(ptr addrspace(1) %out, i32 %i) {
+; CHECK-LABEL: define amdgpu_kernel void @not_promoted_call(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; CHECK-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; CHECK-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; CHECK-NEXT:    store i32 7, ptr addrspace(5) [[P]], align 4
+; CHECK-NEXT:    call void @extern()
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr addrspace(5) [[P]], align 4
+; CHECK-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+; OFF-LABEL: define amdgpu_kernel void @not_promoted_call(
+; OFF-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; OFF-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; OFF-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; OFF-NEXT:    store i32 7, ptr addrspace(5) [[P]], align 4
+; OFF-NEXT:    call void @extern()
+; OFF-NEXT:    [[V:%.*]] = load i32, ptr addrspace(5) [[P]], align 4
+; OFF-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; OFF-NEXT:    ret void
+;
+  %obj = alloca [8 x i32], align 4, addrspace(5)
+  %p = getelementptr i8, ptr addrspace(5) %obj, i32 %i
+  store i32 7, ptr addrspace(5) %p, align 4
+  call void @extern()
+  %v = load i32, ptr addrspace(5) %p, align 4
+  store i32 %v, ptr addrspace(1) %out
+  ret void
+}
+
+; Only accesses the backend implements: an under-aligned one is not one of them.
+define amdgpu_kernel void @not_promoted_misaligned(ptr addrspace(1) %out, i32 %i) {
+; CHECK-LABEL: define amdgpu_kernel void @not_promoted_misaligned(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; CHECK-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(13), !amdgpu.allocated.vgprs [[META0]]
+; CHECK-NEXT:    call void @llvm.amdgcn.vgpr.lifetime.start.p13(ptr addrspace(13) [[OBJ]])
+; CHECK-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(13) [[OBJ]], i32 [[I]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr addrspace(13) [[P]], align 1
+; CHECK-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+; OFF-LABEL: define amdgpu_kernel void @not_promoted_misaligned(
+; OFF-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; OFF-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; OFF-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; OFF-NEXT:    [[V:%.*]] = load i32, ptr addrspace(5) [[P]], align 1
+; OFF-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; OFF-NEXT:    ret void
+;
+  %obj = alloca [8 x i32], align 4, addrspace(5)
+  %p = getelementptr i8, ptr addrspace(5) %obj, i32 %i
+  %v = load i32, ptr addrspace(5) %p, align 1
+  store i32 %v, ptr addrspace(1) %out
+  ret void
+}
+
+; An access the lowering does not implement would only be diagnosed later, so it
+; is rejected here. This one is neither a whole dword nor 8 or 16 bits, despite
+; being sufficiently aligned.
+define amdgpu_kernel void @not_promoted_odd_size(ptr addrspace(1) %out, i32 %i) {
+; CHECK-LABEL: define amdgpu_kernel void @not_promoted_odd_size(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; CHECK-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; CHECK-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; CHECK-NEXT:    [[V:%.*]] = load [3 x i8], ptr addrspace(5) [[P]], align 4
+; CHECK-NEXT:    store [3 x i8] [[V]], ptr addrspace(1) [[OUT]], align 1
+; CHECK-NEXT:    ret void
+;
+; OFF-LABEL: define amdgpu_kernel void @not_promoted_odd_size(
+; OFF-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; OFF-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; OFF-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; OFF-NEXT:    [[V:%.*]] = load [3 x i8], ptr addrspace(5) [[P]], align 4
+; OFF-NEXT:    store [3 x i8] [[V]], ptr addrspace(1) [[OUT]], align 1
+; OFF-NEXT:    ret void
+;
+  %obj = alloca [8 x i32], align 4, addrspace(5)
+  %p = getelementptr i8, ptr addrspace(5) %obj, i32 %i
+  %v = load [3 x i8], ptr addrspace(5) %p, align 4
+  store [3 x i8] %v, ptr addrspace(1) %out
+  ret void
+}
+
+; A volatile access is not a plain indexed register access.
+define amdgpu_kernel void @not_promoted_volatile(ptr addrspace(1) %out, i32 %i) {
+; CHECK-LABEL: define amdgpu_kernel void @not_promoted_volatile(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; CHECK-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; CHECK-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; CHECK-NEXT:    [[V:%.*]] = load volatile i32, ptr addrspace(5) [[P]], align 4
+; CHECK-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+; OFF-LABEL: define amdgpu_kernel void @not_promoted_volatile(
+; OFF-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; OFF-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; OFF-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; OFF-NEXT:    [[V:%.*]] = load volatile i32, ptr addrspace(5) [[P]], align 4
+; OFF-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; OFF-NEXT:    ret void
+;
+  %obj = alloca [8 x i32], align 4, addrspace(5)
+  %p = getelementptr i8, ptr addrspace(5) %obj, i32 %i
+  %v = load volatile i32, ptr addrspace(5) %p, align 4
+  store i32 %v, ptr addrspace(1) %out
+  ret void
+}
+
+declare void @llvm.memcpy.p5.p5.i32(ptr addrspace(5), ptr addrspace(5), i32, i1)
+
+; A copy between two objects is not an indexed access either.
+define amdgpu_kernel void @not_promoted_memcpy(ptr addrspace(1) %out, i32 %i) {
+; CHECK-LABEL: define amdgpu_kernel void @not_promoted_memcpy(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; CHECK-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; CHECK-NEXT:    [[TMP1:%.*]] = call range(i32 0, 1025) i32 @llvm.r600.read.local.size.y()
+; CHECK-NEXT:    [[TMP2:%.*]] = call range(i32 0, 1025) i32 @llvm.r600.read.local.size.z()
+; CHECK-NEXT:    [[TMP3:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.x()
+; CHECK-NEXT:    [[TMP4:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.y()
+; CHECK-NEXT:    [[TMP5:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.z()
+; CHECK-NEXT:    [[TMP6:%.*]] = mul nuw nsw i32 [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[TMP7:%.*]] = mul i32 [[TMP6]], [[TMP3]]
+; CHECK-NEXT:    [[TMP8:%.*]] = mul nuw nsw i32 [[TMP4]], [[TMP2]]
+; CHECK-NEXT:    [[TMP9:%.*]] = add i32 [[TMP7]], [[TMP8]]
+; CHECK-NEXT:    [[TMP10:%.*]] = add i32 [[TMP9]], [[TMP5]]
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds [1024 x [8 x i32]], ptr addrspace(3) @not_promoted_memcpy.other, i32 0, i32 [[TMP10]]
+; CHECK-NEXT:    call void @llvm.memcpy.p5.p3.i32(ptr addrspace(5) [[OBJ]], ptr addrspace(3) [[TMP11]], i32 32, i1 false)
+; CHECK-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr addrspace(5) [[P]], align 4
+; CHECK-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+; OFF-LABEL: define amdgpu_kernel void @not_promoted_memcpy(
+; OFF-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; OFF-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; OFF-NEXT:    [[TMP1:%.*]] = call range(i32 0, 1025) i32 @llvm.r600.read.local.size.y()
+; OFF-NEXT:    [[TMP2:%.*]] = call range(i32 0, 1025) i32 @llvm.r600.read.local.size.z()
+; OFF-NEXT:    [[TMP3:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.x()
+; OFF-NEXT:    [[TMP4:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.y()
+; OFF-NEXT:    [[TMP5:%.*]] = call range(i32 0, 1024) i32 @llvm.amdgcn.workitem.id.z()
+; OFF-NEXT:    [[TMP6:%.*]] = mul nuw nsw i32 [[TMP1]], [[TMP2]]
+; OFF-NEXT:    [[TMP7:%.*]] = mul i32 [[TMP6]], [[TMP3]]
+; OFF-NEXT:    [[TMP8:%.*]] = mul nuw nsw i32 [[TMP4]], [[TMP2]]
+; OFF-NEXT:    [[TMP9:%.*]] = add i32 [[TMP7]], [[TMP8]]
+; OFF-NEXT:    [[TMP10:%.*]] = add i32 [[TMP9]], [[TMP5]]
+; OFF-NEXT:    [[TMP11:%.*]] = getelementptr inbounds [1024 x [8 x i32]], ptr addrspace(3) @not_promoted_memcpy.other, i32 0, i32 [[TMP10]]
+; OFF-NEXT:    call void @llvm.memcpy.p5.p3.i32(ptr addrspace(5) [[OBJ]], ptr addrspace(3) [[TMP11]], i32 32, i1 false)
+; OFF-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; OFF-NEXT:    [[V:%.*]] = load i32, ptr addrspace(5) [[P]], align 4
+; OFF-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; OFF-NEXT:    ret void
+;
+  %obj = alloca [8 x i32], align 4, addrspace(5)
+  %other = alloca [8 x i32], align 4, addrspace(5)
+  call void @llvm.memcpy.p5.p5.i32(ptr addrspace(5) %obj, ptr addrspace(5) %other,
+  i32 32, i1 false)
+  %p = getelementptr i8, ptr addrspace(5) %obj, i32 %i
+  %v = load i32, ptr addrspace(5) %p, align 4
+  store i32 %v, ptr addrspace(1) %out
+  ret void
+}
+
+; Where an alloca qualifies for both, vectorization wins: the two spend the same
+; registers, and a vector needs no indexed access to read an element.
+define amdgpu_kernel void @vector_preferred(ptr addrspace(1) %out, i32 %i) {
+; CHECK-LABEL: define amdgpu_kernel void @vector_preferred(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; CHECK-NEXT:    [[TMP1:%.*]] = freeze <8 x i32> poison
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <8 x i32> [[TMP1]], i32 7, i32 [[I]]
+; CHECK-NEXT:    store i32 7, ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+; OFF-LABEL: define amdgpu_kernel void @vector_preferred(
+; OFF-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; OFF-NEXT:    [[OBJ:%.*]] = freeze <8 x i32> poison
+; OFF-NEXT:    [[TMP1:%.*]] = insertelement <8 x i32> [[OBJ]], i32 7, i32 [[I]]
+; OFF-NEXT:    store i32 7, ptr addrspace(1) [[OUT]], align 4
+; OFF-NEXT:    ret void
+;
+  %obj = alloca [8 x i32], align 4, addrspace(5)
+  %p = getelementptr [8 x i32], ptr addrspace(5) %obj, i32 0, i32 %i
+  store i32 7, ptr addrspace(5) %p
+  %v = load i32, ptr addrspace(5) %p
+  store i32 %v, ptr addrspace(1) %out
+  ret void
+}
+;.
+; CHECK: [[META0]] = !{i32 0, i32 32}
+;.

>From 6a91e9844cf33772b6b7c69ff149f9e62b7a34f9 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Tue, 11 Aug 2026 18:37:27 +0200
Subject: [PATCH 2/2] Carry null operands and intrinsic names across when
 promoting to VGPRs

---
 .../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp |  44 +++++-
 .../AddressSpaceVGPR/as-vgpr-promote.ll       | 145 ++++++++++++++++++
 2 files changed, 187 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index 27be32ee05d40..920af6004fc1a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -1445,14 +1445,54 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVGPR(AllocaAnalysis &AA) const {
 // Move the alloca into the VGPR ("as memory") address space. Pointers into it
 // are the same size in both address spaces, so the pointers derived from it
 // only need their type changed, and allocateVgprs then gives the object its
-// place in that address space.
+// place in that address space. This mirrors what the LDS promotion below does
+// to the same kind of closed set of derived pointers.
 void AMDGPUPromoteAllocaImpl::promoteAllocaToVGPR(AllocaAnalysis &AA) {
   LLVM_DEBUG(dbgs() << "Promoting alloca to VGPRs: " << *AA.Alloca << '\n');
 
   Type *PtrTy = PointerType::get(Mod.getContext(), AMDGPUAS::VGPR);
-  for (Value *Ptr : AA.Pointers)
+  for (Value *Ptr : AA.Pointers) {
     Ptr->mutateType(PtrTy);
 
+    // A select or phi may pick between a pointer into the object and a null
+    // one, which collectAllocaUses allows. Changing the address space of the
+    // result leaves such a constant behind in the old one, so adjust it too.
+    if (auto *SI = dyn_cast<SelectInst>(Ptr)) {
+      for (unsigned I : {1, 2})
+        if (isa<ConstantPointerNull, ConstantAggregateZero>(SI->getOperand(I)))
+          SI->setOperand(I, Constant::getNullValue(PtrTy));
+    } else if (auto *Phi = dyn_cast<PHINode>(Ptr)) {
+      for (unsigned I = 0, E = Phi->getNumIncomingValues(); I != E; ++I)
+        if (isa<ConstantPointerNull, ConstantAggregateZero>(
+                Phi->getIncomingValue(I)))
+          Phi->setIncomingValue(I, Constant::getNullValue(PtrTy));
+    }
+  }
+
+  // An intrinsic overloaded on the pointer type still names the old address
+  // space in its mangled name, which no longer matches the argument it is being
+  // given. Rebuild those, letting the builder derive the name from the types.
+  // The lifetime markers are left alone: allocateVgprs replaces them with the
+  // address-space specific ones below.
+  SmallSetVector<IntrinsicInst *, 4> Rebuild;
+  for (Use *U : AA.Uses) {
+    auto *II = dyn_cast<IntrinsicInst>(U->getUser());
+    if (II && !II->isLifetimeStartOrEnd())
+      Rebuild.insert(II);
+  }
+
+  for (IntrinsicInst *II : Rebuild) {
+    IRBuilder<> B(II);
+    SmallVector<Value *> Args(II->args());
+    Value *New = B.CreateIntrinsic(II->getType(), II->getIntrinsicID(), Args);
+    if (auto *NewCall = dyn_cast<CallInst>(New)) {
+      NewCall->copyMetadata(*II);
+      NewCall->takeName(II);
+    }
+    II->replaceAllUsesWith(New);
+    II->eraseFromParent();
+  }
+
   allocateVgprs(AA);
 }
 
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-promote.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-promote.ll
index c0b15b5f954e7..e79286a838b29 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-promote.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-promote.ll
@@ -250,6 +250,151 @@ define amdgpu_kernel void @not_promoted_memcpy(ptr addrspace(1) %out, i32 %i) {
   ret void
 }
 
+; An intrinsic overloaded on the pointer type has the address space in its
+; mangled name, so moving the object has to rebuild the call, or the name no
+; longer describes the argument.
+define amdgpu_kernel void @memset_object(ptr addrspace(1) %out, i32 %i) {
+; CHECK-LABEL: define amdgpu_kernel void @memset_object(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; CHECK-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(13), !amdgpu.allocated.vgprs [[META0]]
+; CHECK-NEXT:    call void @llvm.amdgcn.vgpr.lifetime.start.p13(ptr addrspace(13) [[OBJ]])
+; CHECK-NEXT:    call void @llvm.memset.p13.i32(ptr addrspace(13) [[OBJ]], i8 0, i32 32, i1 false)
+; CHECK-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(13) [[OBJ]], i32 [[I]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr addrspace(13) [[P]], align 4
+; CHECK-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+; OFF-LABEL: define amdgpu_kernel void @memset_object(
+; OFF-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; OFF-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; OFF-NEXT:    call void @llvm.memset.p5.i32(ptr addrspace(5) [[OBJ]], i8 0, i32 32, i1 false)
+; OFF-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; OFF-NEXT:    [[V:%.*]] = load i32, ptr addrspace(5) [[P]], align 4
+; OFF-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; OFF-NEXT:    ret void
+;
+  %obj = alloca [8 x i32], align 4, addrspace(5)
+  call void @llvm.memset.p5.i32(ptr addrspace(5) %obj, i8 0, i32 32, i1 false)
+  %p = getelementptr i8, ptr addrspace(5) %obj, i32 %i
+  %v = load i32, ptr addrspace(5) %p, align 4
+  store i32 %v, ptr addrspace(1) %out
+  ret void
+}
+
+define amdgpu_kernel void @objectsize_object(ptr addrspace(1) %out, i32 %i) {
+; CHECK-LABEL: define amdgpu_kernel void @objectsize_object(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; CHECK-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(13), !amdgpu.allocated.vgprs [[META0]]
+; CHECK-NEXT:    call void @llvm.amdgcn.vgpr.lifetime.start.p13(ptr addrspace(13) [[OBJ]])
+; CHECK-NEXT:    [[S:%.*]] = call i64 @llvm.objectsize.i64.p13(ptr addrspace(13) [[OBJ]], i1 false, i1 false, i1 false)
+; CHECK-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(13) [[OBJ]], i32 [[I]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr addrspace(13) [[P]], align 4
+; CHECK-NEXT:    [[T:%.*]] = trunc i64 [[S]] to i32
+; CHECK-NEXT:    [[R:%.*]] = add i32 [[V]], [[T]]
+; CHECK-NEXT:    store i32 [[R]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+; OFF-LABEL: define amdgpu_kernel void @objectsize_object(
+; OFF-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]]) {
+; OFF-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; OFF-NEXT:    [[S:%.*]] = call i64 @llvm.objectsize.i64.p5(ptr addrspace(5) [[OBJ]], i1 false, i1 false, i1 false)
+; OFF-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; OFF-NEXT:    [[V:%.*]] = load i32, ptr addrspace(5) [[P]], align 4
+; OFF-NEXT:    [[T:%.*]] = trunc i64 [[S]] to i32
+; OFF-NEXT:    [[R:%.*]] = add i32 [[V]], [[T]]
+; OFF-NEXT:    store i32 [[R]], ptr addrspace(1) [[OUT]], align 4
+; OFF-NEXT:    ret void
+;
+  %obj = alloca [8 x i32], align 4, addrspace(5)
+  %s = call i64 @llvm.objectsize.i64.p5(ptr addrspace(5) %obj, i1 false, i1 false, i1 false)
+  %p = getelementptr i8, ptr addrspace(5) %obj, i32 %i
+  %v = load i32, ptr addrspace(5) %p, align 4
+  %t = trunc i64 %s to i32
+  %r = add i32 %v, %t
+  store i32 %r, ptr addrspace(1) %out
+  ret void
+}
+
+; A select or phi is allowed to pick between a pointer into the object and a
+; null one. Moving the result to another address space has to take the constant
+; with it, or the operands no longer agree and the IR is invalid.
+define amdgpu_kernel void @select_null(ptr addrspace(1) %out, i32 %i, i1 %c) {
+; CHECK-LABEL: define amdgpu_kernel void @select_null(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]], i1 [[C:%.*]]) {
+; CHECK-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(13), !amdgpu.allocated.vgprs [[META0]]
+; CHECK-NEXT:    call void @llvm.amdgcn.vgpr.lifetime.start.p13(ptr addrspace(13) [[OBJ]])
+; CHECK-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(13) [[OBJ]], i32 [[I]]
+; CHECK-NEXT:    [[SEL:%.*]] = select i1 [[C]], ptr addrspace(13) [[P]], ptr addrspace(13) null
+; CHECK-NEXT:    store i32 7, ptr addrspace(13) [[SEL]], align 4
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr addrspace(13) [[SEL]], align 4
+; CHECK-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+; OFF-LABEL: define amdgpu_kernel void @select_null(
+; OFF-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]], i1 [[C:%.*]]) {
+; OFF-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; OFF-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; OFF-NEXT:    [[SEL:%.*]] = select i1 [[C]], ptr addrspace(5) [[P]], ptr addrspace(5) null
+; OFF-NEXT:    store i32 7, ptr addrspace(5) [[SEL]], align 4
+; OFF-NEXT:    [[V:%.*]] = load i32, ptr addrspace(5) [[SEL]], align 4
+; OFF-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; OFF-NEXT:    ret void
+;
+  %obj = alloca [8 x i32], align 4, addrspace(5)
+  %p = getelementptr i8, ptr addrspace(5) %obj, i32 %i
+  %sel = select i1 %c, ptr addrspace(5) %p, ptr addrspace(5) null
+  store i32 7, ptr addrspace(5) %sel, align 4
+  %v = load i32, ptr addrspace(5) %sel, align 4
+  store i32 %v, ptr addrspace(1) %out
+  ret void
+}
+
+define amdgpu_kernel void @phi_null(ptr addrspace(1) %out, i32 %i, i1 %c) {
+; CHECK-LABEL: define amdgpu_kernel void @phi_null(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]], i1 [[C:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(13), !amdgpu.allocated.vgprs [[META0]]
+; CHECK-NEXT:    call void @llvm.amdgcn.vgpr.lifetime.start.p13(ptr addrspace(13) [[OBJ]])
+; CHECK-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(13) [[OBJ]], i32 [[I]]
+; CHECK-NEXT:    br i1 [[C]], label %[[USE:.*]], label %[[OTHER:.*]]
+; CHECK:       [[OTHER]]:
+; CHECK-NEXT:    br label %[[USE]]
+; CHECK:       [[USE]]:
+; CHECK-NEXT:    [[PH:%.*]] = phi ptr addrspace(13) [ [[P]], %[[ENTRY]] ], [ null, %[[OTHER]] ]
+; CHECK-NEXT:    store i32 7, ptr addrspace(13) [[PH]], align 4
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr addrspace(13) [[PH]], align 4
+; CHECK-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; CHECK-NEXT:    ret void
+;
+; OFF-LABEL: define amdgpu_kernel void @phi_null(
+; OFF-SAME: ptr addrspace(1) [[OUT:%.*]], i32 [[I:%.*]], i1 [[C:%.*]]) {
+; OFF-NEXT:  [[ENTRY:.*]]:
+; OFF-NEXT:    [[OBJ:%.*]] = alloca [8 x i32], align 4, addrspace(5)
+; OFF-NEXT:    [[P:%.*]] = getelementptr i8, ptr addrspace(5) [[OBJ]], i32 [[I]]
+; OFF-NEXT:    br i1 [[C]], label %[[USE:.*]], label %[[OTHER:.*]]
+; OFF:       [[OTHER]]:
+; OFF-NEXT:    br label %[[USE]]
+; OFF:       [[USE]]:
+; OFF-NEXT:    [[PH:%.*]] = phi ptr addrspace(5) [ [[P]], %[[ENTRY]] ], [ null, %[[OTHER]] ]
+; OFF-NEXT:    store i32 7, ptr addrspace(5) [[PH]], align 4
+; OFF-NEXT:    [[V:%.*]] = load i32, ptr addrspace(5) [[PH]], align 4
+; OFF-NEXT:    store i32 [[V]], ptr addrspace(1) [[OUT]], align 4
+; OFF-NEXT:    ret void
+;
+entry:
+  %obj = alloca [8 x i32], align 4, addrspace(5)
+  %p = getelementptr i8, ptr addrspace(5) %obj, i32 %i
+  br i1 %c, label %use, label %other
+other:
+  br label %use
+use:
+  %ph = phi ptr addrspace(5) [ %p, %entry ], [ null, %other ]
+  store i32 7, ptr addrspace(5) %ph, align 4
+  %v = load i32, ptr addrspace(5) %ph, align 4
+  store i32 %v, ptr addrspace(1) %out
+  ret void
+}
+
 ; Where an alloca qualifies for both, vectorization wins: the two spend the same
 ; registers, and a vector needs no indexed access to read an element.
 define amdgpu_kernel void @vector_preferred(ptr addrspace(1) %out, i32 %i) {



More information about the llvm-branch-commits mailing list