[llvm] [AMDGPU] Lower loads and stores for address space 13 (PR #209541)

Gheorghe-Teodor Bercea via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 30 05:46:30 PDT 2026


https://github.com/doru1004 updated https://github.com/llvm/llvm-project/pull/209541

>From 4c8877ed109e3c88b32fff0a9f1c7e8d81ab413e Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Tue, 14 Jul 2026 00:06:53 -0500
Subject: [PATCH 01/35] Lower loads and stores for address space 13

---
 llvm/lib/Target/AMDGPU/AMDGPU.h               |   9 +
 .../lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp | 108 ++++
 llvm/lib/Target/AMDGPU/AMDGPUGISel.td         |   3 +
 .../AMDGPU/AMDGPUInstructionSelector.cpp      |  27 +
 .../Target/AMDGPU/AMDGPUInstructionSelector.h |   1 +
 .../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp | 100 ++++
 .../Target/AMDGPU/AMDGPULowerVGPREncoding.cpp |  99 +++-
 .../lib/Target/AMDGPU/AMDGPUMachineInstrs.cpp |  55 +++
 llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h  |  71 +++
 llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def |   1 +
 .../AMDGPU/AMDGPURegBankLegalizeRules.cpp     |   7 +
 .../Target/AMDGPU/AMDGPURegisterBankInfo.cpp  |  15 +
 .../Target/AMDGPU/AMDGPUSearchableTables.td   |  17 +
 .../lib/Target/AMDGPU/AMDGPUTargetMachine.cpp |   9 +
 .../AMDGPU/AMDGPUTargetTransformInfo.cpp      |   7 +-
 llvm/lib/Target/AMDGPU/CMakeLists.txt         |   2 +
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     | 117 +++++
 llvm/lib/Target/AMDGPU/SIISelLowering.h       |   3 +
 llvm/lib/Target/AMDGPU/SIInstrInfo.cpp        |  73 ++-
 llvm/lib/Target/AMDGPU/SIInstrInfo.td         |  11 +
 llvm/lib/Target/AMDGPU/SIInstructions.td      | 100 ++++
 .../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp    |  11 +
 llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h |  13 +
 .../AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll  | 384 +++++++++++++++
 .../AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll   | 461 ++++++++++++++++++
 .../AddressSpaceVGPR/as-vgpr-divergent.ll     | 130 +++++
 .../AddressSpaceVGPR/as-vgpr-inttoptr.ll      |  48 ++
 .../AddressSpaceVGPR/as-vgpr-unsupported.ll   |  31 ++
 llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll  |   3 +
 llvm/test/CodeGen/AMDGPU/llc-pipeline.ll      |   5 +
 30 files changed, 1897 insertions(+), 24 deletions(-)
 create mode 100644 llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
 create mode 100644 llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.cpp
 create mode 100644 llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-inttoptr.ll
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.h b/llvm/lib/Target/AMDGPU/AMDGPU.h
index 809540bd05b45..87ebce3901845 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.h
@@ -288,6 +288,9 @@ extern char &AMDGPURegBankLegalizeLegacyID;
 void initializeAMDGPUMarkLastScratchLoadLegacyPass(PassRegistry &);
 extern char &AMDGPUMarkLastScratchLoadID;
 
+void initializeAMDGPUAssignIdxToM0LegacyPass(PassRegistry &);
+extern char &AMDGPUAssignIdxToM0ID;
+
 void initializeSILowerSGPRSpillsLegacyPass(PassRegistry &);
 extern char &SILowerSGPRSpillsLegacyID;
 
@@ -504,6 +507,12 @@ class AMDGPUMarkLastScratchLoadPass
                         MachineFunctionAnalysisManager &AM);
 };
 
+class AMDGPUAssignIdxToM0Pass : public PassInfoMixin<AMDGPUAssignIdxToM0Pass> {
+public:
+  PreservedAnalyses run(MachineFunction &MF,
+                        MachineFunctionAnalysisManager &MFAM);
+};
+
 class SIInsertWaitcntsPass
     : public RequiredPassInfoMixin<SIInsertWaitcntsPass> {
 public:
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
new file mode 100644
index 0000000000000..8c5de37eb46e5
--- /dev/null
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
@@ -0,0 +1,108 @@
+//===- AMDGPUAssignIdxToM0.cpp - Copy VGPR-memory indices to M0 ----------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+/// \file
+/// Copy the register index of a VGPR "as memory" (address space 13)
+/// V_LOAD_IDX / V_STORE_IDX pseudo into M0, which V_MOVREL[SD] reads when the
+/// pseudo is lowered (see AMDGPULowerVGPREncoding). This runs before register
+/// allocation so the copy to M0 is inserted while the index is still virtual.
+//
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPU.h"
+#include "AMDGPUMachineInstrs.h"
+#include "GCNSubtarget.h"
+#include "SIInstrInfo.h"
+#include "llvm/CodeGen/MachineFunctionPass.h"
+#include "llvm/CodeGen/MachineInstrBuilder.h"
+#include "llvm/CodeGen/MachinePassManager.h"
+#include "llvm/InitializePasses.h"
+
+using namespace llvm;
+
+#define DEBUG_TYPE "amdgpu-assign-idx-to-m0"
+
+static bool assignIdxToM0(MachineFunction &MF) {
+  const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
+  if (!ST.hasMovrel())
+    return false;
+
+  const SIInstrInfo *TII = ST.getInstrInfo();
+
+  bool Changed = false;
+  for (MachineBasicBlock &MBB : MF) {
+    for (MachineInstr &MI : MBB) {
+      auto *LdSt = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI);
+      if (!LdSt)
+        continue;
+
+      MachineOperand &IdxOp = LdSt->getIdxOp();
+      if (!IdxOp.isReg())
+        continue;
+
+      assert(!MI.isBundled());
+
+      // Remove the implicit-def $m0 that instruction selection added (to pin a
+      // divergent access inside its waterfall loop); M0 is written for real
+      // below.
+      int DefIdx = MI.findRegisterDefOperandIdx(AMDGPU::M0, /*TRI=*/nullptr);
+      assert(DefIdx >= 0);
+      MI.removeOperand(DefIdx);
+
+      // Add a copy from the index register to M0 and rewrite MI to read M0.
+      BuildMI(MBB, &MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY), AMDGPU::M0)
+          .add(IdxOp);
+      IdxOp.setReg(AMDGPU::M0);
+      IdxOp.setIsKill();
+      Changed = true;
+    }
+  }
+
+  return Changed;
+}
+
+namespace {
+
+class AMDGPUAssignIdxToM0Legacy : public MachineFunctionPass {
+public:
+  static char ID;
+
+  AMDGPUAssignIdxToM0Legacy() : MachineFunctionPass(ID) {}
+
+  bool runOnMachineFunction(MachineFunction &MF) override {
+    if (skipFunction(MF.getFunction()))
+      return false;
+    return assignIdxToM0(MF);
+  }
+
+  void getAnalysisUsage(AnalysisUsage &AU) const override {
+    AU.setPreservesCFG();
+    MachineFunctionPass::getAnalysisUsage(AU);
+  }
+
+  StringRef getPassName() const override { return "AMDGPU Assign Idx To M0"; }
+};
+
+} // end anonymous namespace
+
+PreservedAnalyses
+AMDGPUAssignIdxToM0Pass::run(MachineFunction &MF,
+                             MachineFunctionAnalysisManager &MFAM) {
+  if (!assignIdxToM0(MF))
+    return PreservedAnalyses::all();
+  auto PA = getMachineFunctionPassPreservedAnalyses();
+  PA.preserveSet<CFGAnalyses>();
+  return PA;
+}
+
+char AMDGPUAssignIdxToM0Legacy::ID = 0;
+
+char &llvm::AMDGPUAssignIdxToM0ID = AMDGPUAssignIdxToM0Legacy::ID;
+
+INITIALIZE_PASS(AMDGPUAssignIdxToM0Legacy, DEBUG_TYPE,
+                "AMDGPU Assign Idx To M0", false, false)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUGISel.td b/llvm/lib/Target/AMDGPU/AMDGPUGISel.td
index a1c4af109b44a..f55bc70ca31f4 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUGISel.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPUGISel.td
@@ -290,6 +290,9 @@ def : GINodeEquiv<G_AMDGPU_FMINIMUM3, AMDGPUfminimum3>;
 
 def : GINodeEquiv<G_AMDGPU_CLAMP, AMDGPUclamp>;
 
+def : GINodeEquiv<G_AMDGPU_REG_LOAD,  SIreg_load>;
+def : GINodeEquiv<G_AMDGPU_REG_STORE, SIreg_store>;
+
 def : GINodeEquiv<G_AMDGPU_ATOMIC_CMPXCHG, AMDGPUatomic_cmp_swap>;
 def : GINodeEquiv<G_AMDGPU_BUFFER_LOAD, SIbuffer_load>;
 def : GINodeEquiv<G_AMDGPU_BUFFER_LOAD_USHORT, SIbuffer_load_ushort>;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
index f06837ff13869..afc1607ee1b0d 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
@@ -15,6 +15,7 @@
 #include "AMDGPU.h"
 #include "AMDGPUGlobalISelUtils.h"
 #include "AMDGPUInstrInfo.h"
+#include "AMDGPUMachineInstrs.h"
 #include "AMDGPURegisterBankInfo.h"
 #include "SIMachineFunctionInfo.h"
 #include "Utils/AMDGPUBaseInfo.h"
@@ -4520,6 +4521,29 @@ bool AMDGPUInstructionSelector::selectStackRestore(MachineInstr &MI) const {
   return true;
 }
 
+bool AMDGPUInstructionSelector::selectRegLoadStore(MachineInstr &I) const {
+  // Remember where the selected machine instruction will land.
+  MachineBasicBlock::iterator II = std::next(I.getIterator());
+
+  if (!selectImpl(I, *CoverageInfo))
+    return false;
+
+  // On a movrel subtarget the selected V_LOAD_IDX / V_STORE_IDX expands to an
+  // M0-relative move (see AMDGPUAssignIdxToM0 and AMDGPULowerVGPREncoding). Add
+  // an implicit-def of $m0: it records that the eventual move clobbers M0, and
+  // - because an instruction defining a physical register is not hoisted/sunk -
+  // keeps a divergent access pinned inside its waterfall loop.
+  // AMDGPUAssignIdxToM0 removes it when it writes M0 for real.
+  if (!Subtarget->hasMovrel())
+    return true;
+
+  auto *LdStIdx = cast<AMDGPUMI::VLoadStoreIdxInst>(&*std::prev(II));
+  if (LdStIdx->getIdxOp().isReg())
+    LdStIdx->addOperand(MachineOperand::CreateReg(AMDGPU::M0, /*isDef=*/true,
+                                                  /*isImp=*/true));
+  return true;
+}
+
 bool AMDGPUInstructionSelector::select(MachineInstr &I) {
 
   if (!I.isPreISelOpcode()) {
@@ -4655,6 +4679,9 @@ bool AMDGPUInstructionSelector::select(MachineInstr &I) {
   case AMDGPU::G_AMDGPU_BVH_INTERSECT_RAY:
   case AMDGPU::G_AMDGPU_BVH8_INTERSECT_RAY:
     return selectBVHIntersectRayIntrinsic(I);
+  case AMDGPU::G_AMDGPU_REG_LOAD:
+  case AMDGPU::G_AMDGPU_REG_STORE:
+    return selectRegLoadStore(I);
   case AMDGPU::G_SBFX:
   case AMDGPU::G_UBFX:
     return selectG_SBFX_UBFX(I);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.h b/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.h
index 71e815bf122d8..ab4084472da36 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.h
@@ -91,6 +91,7 @@ class AMDGPUInstructionSelector final : public InstructionSelector {
   bool selectCOPY_VCC_SCC(MachineInstr &I) const;
   bool selectReadAnyLane(MachineInstr &I) const;
   bool selectPHI(MachineInstr &I) const;
+  bool selectRegLoadStore(MachineInstr &I) const;
   bool selectG_TRUNC(MachineInstr &I) const;
   bool selectG_SZA_EXT(MachineInstr &I) const;
   bool selectG_FPEXT(MachineInstr &I) const;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index e759f6eb64860..778f6efe6ffce 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -16,6 +16,7 @@
 #include "AMDGPU.h"
 #include "AMDGPUGlobalISelUtils.h"
 #include "AMDGPUInstrInfo.h"
+#include "AMDGPUMachineInstrs.h"
 #include "AMDGPUMemoryUtils.h"
 #include "AMDGPUTargetMachine.h"
 #include "SIInstrInfo.h"
@@ -432,6 +433,8 @@ static unsigned maxSizeForAddrSpace(const GCNSubtarget &ST, unsigned AS,
     // register bank/uniformity and if the memory is invariant or not written in a
     // kernel.
     return IsLoad ? 512 : 128;
+  case AMDGPUAS::VGPR:
+    return 1024;
   default:
     // FIXME: Flat addresses may contextually need to be split to 32-bit parts
     // if they may alias scratch depending on the subtarget.  This needs to be
@@ -548,10 +551,23 @@ static bool loadStoreBitcastWorkaround(const LLT Ty) {
 
 static bool isLoadStoreLegal(const GCNSubtarget &ST, const LegalityQuery &Query) {
   const LLT Ty = Query.Types[0];
+  // VGPR ("as memory") accesses are never plain-legal; they are custom-lowered
+  // to G_AMDGPU_REG_LOAD/STORE.
+  if (Query.Types[1].getAddressSpace() == AMDGPUAS::VGPR)
+    return false;
   return isRegisterType(ST, Ty) && isLoadStoreSizeLegal(ST, Query) &&
          !hasBufferRsrcWorkaround(Ty) && !loadStoreBitcastWorkaround(Ty);
 }
 
+// Whether the VGPR ("as memory") load/store lowering handles a MemSize-bit
+// memory access producing/consuming a ValSize-bit value. Only whole-dword
+// accesses (those with a matching V_LOAD_IDX/V_STORE_IDX pseudo) are supported
+// for now; sub-dword (8/16-bit) support lands later.
+static bool isVGPRLoadStoreSizeSupported(unsigned MemSize, unsigned ValSize) {
+  return MemSize == ValSize &&
+         AMDGPUMI::VLoadIdxInst::tryGetOpcodeForBitWidth(MemSize) != -1;
+}
+
 /// Return true if a load or store of the type should be lowered with a bitcast
 /// to a different type.
 static bool shouldBitcastLoadStoreType(const GCNSubtarget &ST, const LLT Ty,
@@ -1677,6 +1693,14 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
     // inserting addrspacecasts.
     Actions.customIf(typeIs(1, Constant32Ptr));
 
+    // VGPR ("as memory") accesses are custom-lowered to the legal
+    // G_AMDGPU_REG_LOAD/STORE target instructions. Always take the custom path
+    // so an unsupported (e.g. sub-dword) access is diagnosed cleanly rather
+    // than failing to legalize.
+    Actions.customIf([=](const LegalityQuery &Query) -> bool {
+      return Query.Types[1].getAddressSpace() == AMDGPUAS::VGPR;
+    });
+
     // Turn any illegal element vectors into something easier to deal
     // with. These will ultimately produce 32-bit scalar shifts to extract the
     // parts anyway.
@@ -3477,6 +3501,75 @@ static LLT widenToNextPowerOf2(LLT Ty) {
   return Ty.changeElementSize(PowerOf2Ceil(Ty.getSizeInBits()));
 }
 
+/// Lower a whole-dword G_LOAD / G_STORE on AMDGPUAS::VGPR into a legal
+/// G_AMDGPU_REG_LOAD / G_AMDGPU_REG_STORE indexed by the pointer's dword offset
+/// (pointer >> 2). Parallels the SelectionDAG LowerLoadStoreVGPR.
+static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
+  MachineIRBuilder &B = Helper.MIRBuilder;
+  MachineRegisterInfo &MRI = *B.getMRI();
+  MachineMemOperand &MMO = **MI.memoperands_begin();
+
+  const bool IsStore = MI.getOpcode() == AMDGPU::G_STORE;
+  Register ValReg = MI.getOperand(0).getReg();
+  Register PtrReg = MI.getOperand(1).getReg();
+
+  const LLT ValTy = MRI.getType(ValReg);
+  const unsigned ValSize = ValTy.getSizeInBits();
+  // The GISel selection patterns for the indexed pseudos - and for the shift /
+  // readfirstlane that compute the index - match the extended integer LLT, so
+  // build the dword index (and the normalized register value below) with
+  // integer types rather than plain scalars.
+  const LLT I32 = LLT::integer(32);
+
+  // Only whole-dword, non-extending/non-truncating accesses are implemented.
+  // Reject anything else with a diagnostic instead of failing to legalize
+  // (sub-dword support lands in a later change).
+  if (!isVGPRLoadStoreSizeSupported(MMO.getMemoryType().getSizeInBits(),
+                                    ValSize)) {
+    const Function &F = B.getMF().getFunction();
+    F.getContext().diagnose(DiagnosticInfoUnsupported(
+        F,
+        "unsupported access of VGPR 'as memory' address space (13); only "
+        "whole-dword loads and stores are implemented",
+        MI.getDebugLoc()));
+    if (!IsStore)
+      B.buildUndef(ValReg);
+    MI.eraseFromParent();
+    return true;
+  }
+
+  const auto PtrAsInt = B.buildPtrToInt(I32, PtrReg);
+  auto Two = B.buildConstant(I32, 2);
+  const auto Index = B.buildLShr(I32, PtrAsInt, Two);
+
+  // Normalize the value to i32 / <N x i32> so a selection pattern always
+  // exists (e.g. for v4i8).
+  LLT RegTy = ValTy;
+  if (ValTy.getScalarSizeInBits() != 32) {
+    unsigned NumDwords = ValSize / 32;
+    RegTy = NumDwords == 1 ? I32 : LLT::fixed_vector(NumDwords, I32);
+  }
+
+  if (IsStore) {
+    Register Value = ValReg;
+    if (RegTy != ValTy)
+      Value = B.buildBitcast(RegTy, Value).getReg(0);
+    B.buildInstr(AMDGPU::G_AMDGPU_REG_STORE, {}, {Value, Index.getReg(0)})
+        .addMemOperand(&MMO);
+  } else if (RegTy == ValTy) {
+    B.buildInstr(AMDGPU::G_AMDGPU_REG_LOAD, {ValReg}, {Index.getReg(0)})
+        .addMemOperand(&MMO);
+  } else {
+    const auto Result =
+        B.buildInstr(AMDGPU::G_AMDGPU_REG_LOAD, {RegTy}, {Index.getReg(0)})
+            .addMemOperand(&MMO);
+    B.buildBitcast(ValReg, Result);
+  }
+
+  MI.eraseFromParent();
+  return true;
+}
+
 bool AMDGPULegalizerInfo::legalizeLoad(LegalizerHelper &Helper,
                                        MachineInstr &MI) const {
   MachineIRBuilder &B = Helper.MIRBuilder;
@@ -3487,6 +3580,9 @@ bool AMDGPULegalizerInfo::legalizeLoad(LegalizerHelper &Helper,
   LLT PtrTy = MRI.getType(PtrReg);
   unsigned AddrSpace = PtrTy.getAddressSpace();
 
+  if (AddrSpace == AMDGPUAS::VGPR && MI.getOpcode() == AMDGPU::G_LOAD)
+    return lowerLoadStoreVGPR(Helper, MI);
+
   if (AddrSpace == AMDGPUAS::CONSTANT_ADDRESS_32BIT) {
     LLT ConstPtr = LLT::pointer(AMDGPUAS::CONSTANT_ADDRESS, 64);
     auto Cast = B.buildAddrSpaceCast(ConstPtr, PtrReg);
@@ -3575,6 +3671,10 @@ bool AMDGPULegalizerInfo::legalizeStore(LegalizerHelper &Helper,
   Register DataReg = MI.getOperand(0).getReg();
   LLT DataTy = MRI.getType(DataReg);
 
+  if (MRI.getType(MI.getOperand(1).getReg()).getAddressSpace() ==
+      AMDGPUAS::VGPR)
+    return lowerLoadStoreVGPR(Helper, MI);
+
   if (hasBufferRsrcWorkaround(DataTy)) {
     Observer.changingInstr(MI);
     castBufferRsrcArgToV4I32(MI, B, 0);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
index 75c3dd3b1de09..4ae470fac3386 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
@@ -42,9 +42,11 @@
 
 #include "AMDGPULowerVGPREncoding.h"
 #include "AMDGPU.h"
+#include "AMDGPUMachineInstrs.h"
 #include "GCNSubtarget.h"
 #include "SIDefines.h"
 #include "SIInstrInfo.h"
+#include "SIMachineFunctionInfo.h"
 #include "llvm/ADT/bit.h"
 #include "llvm/CodeGen/MachineBasicBlock.h"
 #include "llvm/Support/Debug.h"
@@ -132,6 +134,7 @@ class AMDGPULowerVGPREncoding {
   bool run(MachineFunction &MF);
 
 private:
+  const GCNSubtarget *ST;
   const SIInstrInfo *TII;
   const SIRegisterInfo *TRI;
 
@@ -177,6 +180,13 @@ class AMDGPULowerVGPREncoding {
   /// Handle single \p MI. \return true if changed.
   bool runOnMachineInstr(MachineInstr &MI);
 
+  /// Lower a VGPR "as memory" (address space 13) indexed load/store pseudo
+  /// (V_LOAD_IDX_B<N> / V_STORE_IDX_B<N>) into a sequence of M0-relative moves
+  /// (v_movrels_b32 for loads, v_movreld_b32 for stores) over the wave's vector
+  /// registers. M0 must already hold the dword index (see AMDGPUAssignIdxToM0).
+  /// This replaces the pseudo, which is erased.
+  void lowerLoadStoreIdx(MachineInstr &MI);
+
   /// Compute the mode for a single \p MI given \p Ops operands
   /// bit mapping. Optionally takes second array \p Ops2 for VOPD.
   /// If provided and an operand from \p Ops is not a VGPR, then \p Ops2
@@ -379,6 +389,62 @@ bool AMDGPULowerVGPREncoding::runOnMachineInstr(MachineInstr &MI) {
   return false;
 }
 
+void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
+  auto &LdSt = cast<AMDGPUMI::VLoadStoreIdxInst>(MI);
+  MachineBasicBlock &BB = *MI.getParent();
+  const DebugLoc &DL = MI.getDebugLoc();
+  const bool IsStore = LdSt.mayStore();
+
+  // $data is operand 0 of both the load (def) and store (use) pseudos; M0
+  // already holds the dword index (see AMDGPUAssignIdxToM0).
+  Register Data = LdSt.getDataOp().getReg();
+  unsigned Offset = LdSt.getOffsetOp().getImm();
+  unsigned NumDwords = LdSt.getBitWidth() / 32;
+
+  // A statically out-of-range dword offset would fold into a base VGPR outside
+  // the addressable register file - an out-of-bounds access of the VGPR
+  // "as memory" (address space 13) region. When the whole file is addressable
+  // the index is allowed to wrap; otherwise it must stay in range.
+#ifndef NDEBUG
+  unsigned NumAddressableVGPRs = ST->getAddressableNumVGPRs(
+      MI.getMF()->getInfo<SIMachineFunctionInfo>()->getDynamicVGPRBlockSize());
+  bool AllowOffsetWrap = NumAddressableVGPRs == ST->getTotalNumVGPRs();
+  assert((AllowOffsetWrap || Offset + NumDwords <= NumAddressableVGPRs) &&
+         "out of bounds VGPR 'as memory' (address space 13) access");
+#endif
+
+  unsigned Opcode =
+      IsStore ? AMDGPU::V_MOVRELD_B32_e32 : AMDGPU::V_MOVRELS_B32_e32;
+
+  // The dword index is (M0 + $offset). Fold $offset into the base register so
+  // each dword i reads/writes VGPR($offset + i) relative to M0.
+  for (unsigned i = 0; i < NumDwords; ++i) {
+    Register Base = AMDGPU::VGPR0 + Offset + i;
+    Register Sub = Data;
+    if (NumDwords != 1)
+      Sub = TRI->getSubReg(Data, TRI->getSubRegFromChannel(i));
+
+    MachineInstr *Mov;
+    if (IsStore)
+      Mov = BuildMI(BB, MI, DL, TII->get(Opcode))
+                .addReg(Base, RegState::Undef)
+                .addReg(Sub)
+                .getInstr();
+    else
+      Mov = BuildMI(BB, MI, DL, TII->get(Opcode), Sub)
+                .addReg(Base, RegState::Undef)
+                .getInstr();
+
+    // On subtargets with more than 256 addressable VGPRs the referenced
+    // register may need high address bits; reuse the S_SET_VGPR_MSB machinery
+    // to encode them. This is a no-op on movrel-only (<=256 VGPR) subtargets.
+    if (ST->has1024AddressableVGPRs())
+      runOnMachineInstr(*Mov);
+  }
+
+  MI.eraseFromParent();
+}
+
 MachineBasicBlock::instr_iterator
 AMDGPULowerVGPREncoding::handleClause(MachineBasicBlock::instr_iterator I) {
   if (!ClauseRemaining)
@@ -570,12 +636,14 @@ bool AMDGPULowerVGPREncoding::handleSetregMode(MachineInstr &MI) {
 }
 
 bool AMDGPULowerVGPREncoding::run(MachineFunction &MF) {
-  const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
-  if (!ST.has1024AddressableVGPRs())
-    return false;
+  ST = &MF.getSubtarget<GCNSubtarget>();
+  TII = ST->getInstrInfo();
+  TRI = ST->getRegisterInfo();
 
-  TII = ST.getInstrInfo();
-  TRI = ST.getRegisterInfo();
+  // The S_SET_VGPR_MSB encoding is only required on subtargets with more than
+  // 256 addressable VGPRs (gfx1250). On movrel-only subtargets the pass still
+  // runs, but only to lower the VGPR "as memory" indexed load/store pseudos.
+  const bool LowerVGPRMSBs = ST->has1024AddressableVGPRs();
 
   LLVM_DEBUG(dbgs() << "*** AMDGPULowerVGPREncoding on " << MF.getName()
                     << " ***\n");
@@ -592,6 +660,19 @@ bool AMDGPULowerVGPREncoding::run(MachineFunction &MF) {
                       << ":\n");
 
     for (auto &MI : llvm::make_early_inc_range(MBB.instrs())) {
+      // Lower VGPR "as memory" indexed load/store pseudos on any subtarget that
+      // reaches this pass (movrel-only or gfx1250). This replaces the pseudo.
+      if (isa<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
+        lowerLoadStoreIdx(MI);
+        Changed = true;
+        continue;
+      }
+
+      // The remaining work only inserts VGPR MSB encoding, which is unnecessary
+      // on movrel-only subtargets.
+      if (!LowerVGPRMSBs)
+        continue;
+
       if (MI.isMetaInstruction())
         continue;
 
@@ -624,7 +705,7 @@ bool AMDGPULowerVGPREncoding::run(MachineFunction &MF) {
       }
 
       if (MI.getOpcode() == AMDGPU::S_SETREG_IMM32_B32 &&
-          ST.hasSetregVGPRMSBFixup()) {
+          ST->hasSetregVGPRMSBFixup()) {
         Changed |= handleSetregMode(MI);
         continue;
       }
@@ -648,8 +729,10 @@ bool AMDGPULowerVGPREncoding::run(MachineFunction &MF) {
     }
 
     // Reset the mode if we are falling through.
-    LLVM_DEBUG(dbgs() << "  end of BB, resetting mode\n");
-    resetMode(MBB.instr_end());
+    if (LowerVGPRMSBs) {
+      LLVM_DEBUG(dbgs() << "  end of BB, resetting mode\n");
+      resetMode(MBB.instr_end());
+    }
   }
 
   return Changed;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.cpp
new file mode 100644
index 0000000000000..697c63e2079d1
--- /dev/null
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.cpp
@@ -0,0 +1,55 @@
+//===-- AMDGPUMachineInstrs.cpp -*- C++ -*---------------------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+/// Convenience wrappers and helpers for AMDGPU-specific machine instructions.
+//
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPUMachineInstrs.h"
+#include "Utils/AMDGPUBaseInfo.h"
+#include "llvm/CodeGen/MachineInstr.h"
+#include "llvm/Support/ErrorHandling.h"
+
+using namespace llvm;
+using namespace AMDGPUMI;
+
+unsigned VLoadStoreIdxInst::getBitWidth() const {
+  const AMDGPU::VLdStIdxOpcodeInfo *Info =
+      AMDGPU::getVLdStIdxOpcodeInfoByOpcode(getOpcode());
+  if (!Info)
+    llvm_unreachable("unsupported V_LOAD/STORE_IDX opcode");
+  return Info->BitWidth;
+}
+
+int VLoadIdxInst::tryGetOpcodeForBitWidth(unsigned Bits) {
+  const AMDGPU::VLdStIdxOpcodeInfo *Info =
+      AMDGPU::getVLdStIdxOpcodeInfoByKey(Bits, /*IsStore=*/false);
+  if (!Info)
+    return -1;
+  return Info->Opcode;
+}
+
+unsigned VLoadIdxInst::getOpcodeForBitWidth(unsigned Bits) {
+  int Opcode = tryGetOpcodeForBitWidth(Bits);
+  assert(Opcode != -1);
+  return Opcode;
+}
+
+int VStoreIdxInst::tryGetOpcodeForBitWidth(unsigned Bits) {
+  const AMDGPU::VLdStIdxOpcodeInfo *Info =
+      AMDGPU::getVLdStIdxOpcodeInfoByKey(Bits, /*IsStore=*/true);
+  if (!Info)
+    return -1;
+  return Info->Opcode;
+}
+
+unsigned VStoreIdxInst::getOpcodeForBitWidth(unsigned Bits) {
+  int Opcode = tryGetOpcodeForBitWidth(Bits);
+  assert(Opcode != -1);
+  return Opcode;
+}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h b/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h
new file mode 100644
index 0000000000000..24e24aab53152
--- /dev/null
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h
@@ -0,0 +1,71 @@
+//===-- AMDGPUMachineInstrs.h -*- C++ -*-----------------------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+/// Convenience wrappers and helpers for AMDGPU-specific machine instructions.
+//
+//===----------------------------------------------------------------------===//
+
+#ifndef LLVM_LIB_TARGET_AMDGPU_AMDGPUMACHINEINSTRS_H
+#define LLVM_LIB_TARGET_AMDGPU_AMDGPUMACHINEINSTRS_H
+
+#include "SIInstrInfo.h"
+#include "Utils/AMDGPUBaseInfo.h"
+#include "llvm/CodeGen/MachineInstr.h"
+
+namespace llvm {
+namespace AMDGPUMI {
+
+// Wrapper for the whole-dword VGPR "as memory" (address space 13) indexed
+// load/store pseudos (V_LOAD_IDX_B<N> / V_STORE_IDX_B<N>). Operand layout:
+//   load:  (outs data), (ins idx, offset)
+//   store: (outs),      (ins data, idx, offset)
+// so data/idx/offset are always operands 0/1/2.
+class VLoadStoreIdxInst : public MachineInstr {
+public:
+  MachineOperand &getDataOp() { return getOperand(0); }
+  MachineOperand &getIdxOp() { return getOperand(1); }
+  MachineOperand &getOffsetOp() { return getOperand(2); }
+  const MachineOperand &getDataOp() const { return getOperand(0); }
+  const MachineOperand &getIdxOp() const { return getOperand(1); }
+  const MachineOperand &getOffsetOp() const { return getOperand(2); }
+
+  unsigned getBitWidth() const;
+
+  static bool classof(const MachineInstr *MI) {
+    return AMDGPU::getVLdStIdxOpcodeInfoByOpcode(MI->getOpcode()) != nullptr;
+  }
+};
+
+class VLoadIdxInst : public VLoadStoreIdxInst {
+public:
+  static int tryGetOpcodeForBitWidth(unsigned Bits);
+  static unsigned getOpcodeForBitWidth(unsigned Bits);
+
+  static bool classof(const MachineInstr *MI) {
+    const AMDGPU::VLdStIdxOpcodeInfo *Info =
+        AMDGPU::getVLdStIdxOpcodeInfoByOpcode(MI->getOpcode());
+    return Info && !Info->IsStore;
+  }
+};
+
+class VStoreIdxInst : public VLoadStoreIdxInst {
+public:
+  static int tryGetOpcodeForBitWidth(unsigned Bits);
+  static unsigned getOpcodeForBitWidth(unsigned Bits);
+
+  static bool classof(const MachineInstr *MI) {
+    const AMDGPU::VLdStIdxOpcodeInfo *Info =
+        AMDGPU::getVLdStIdxOpcodeInfoByOpcode(MI->getOpcode());
+    return Info && Info->IsStore;
+  }
+};
+
+} // end namespace AMDGPUMI
+} // end namespace llvm
+
+#endif // LLVM_LIB_TARGET_AMDGPU_AMDGPUMACHINEINSTRS_H
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def b/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def
index 372d5f5acab21..e7a4435db36c8 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def
@@ -116,6 +116,7 @@ MACHINE_FUNCTION_ANALYSIS("amdgpu-next-use-analysis", AMDGPUNextUseAnalysisPass(
 #define MACHINE_FUNCTION_PASS(NAME, CREATE_PASS)
 #endif
 MACHINE_FUNCTION_PASS("amdgpu-asm-printer", AMDGPUAsmPrinterPass())
+MACHINE_FUNCTION_PASS("amdgpu-assign-idx-to-m0", AMDGPUAssignIdxToM0Pass())
 MACHINE_FUNCTION_PASS("amdgpu-global-isel-divergence-lowering",
                       AMDGPUGlobalISelDivergenceLoweringPass())
 MACHINE_FUNCTION_PASS("amdgpu-insert-delay-alu", AMDGPUInsertDelayAluPass())
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp b/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
index 1d92bca3e12ab..2b646b66af132 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
@@ -1393,6 +1393,13 @@ RegBankLegalizeRules::RegBankLegalizeRules(const GCNSubtarget &_ST,
             {{VgprV2S16},
              {VgprV2S16, SgprV4S32_WF, Vgpr32, Vgpr32, Sgpr32_WF}}});
 
+  // VGPR ("as memory") indexed load/store: the data is a VGPR value of any
+  // register class; the dword index is made uniform (waterfall) as an SGPR.
+  addRulesForGOpcs({G_AMDGPU_REG_LOAD}).Any({{BRC}, {{VgprBRC}, {Sgpr32_WF}}});
+
+  addRulesForGOpcs({G_AMDGPU_REG_STORE})
+      .Any({{BRC}, {{}, {VgprBRC, Sgpr32_WF}}});
+
   addRulesForGOpcs({G_PTR_ADD})
       .Any({{UniPtr32}, {{SgprPtr32}, {SgprPtr32, Sgpr32}}})
       .Any({{DivPtr32}, {{VgprPtr32}, {VgprPtr32, Vgpr32}}})
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
index 91bd0d006cc64..dec1dbed0f22e 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
@@ -3088,6 +3088,14 @@ void AMDGPURegisterBankInfo::applyMappingImpl(
 
     return;
   }
+  case AMDGPU::G_AMDGPU_REG_LOAD:
+  case AMDGPU::G_AMDGPU_REG_STORE: {
+    // The dword index (operand 1) must be uniform; a divergent index needs a
+    // waterfall loop.
+    applyDefaultMapping(OpdMapper);
+    executeInWaterfallLoop(B, MI, {1});
+    return;
+  }
   case AMDGPU::G_AMDGPU_BUFFER_LOAD:
   case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT:
   case AMDGPU::G_AMDGPU_BUFFER_LOAD_SSHORT:
@@ -4488,6 +4496,13 @@ AMDGPURegisterBankInfo::getInstrMapping(const MachineInstr &MI) const {
     }
     break;
   }
+  case AMDGPU::G_AMDGPU_REG_LOAD:
+  case AMDGPU::G_AMDGPU_REG_STORE: {
+    // data/result is a VGPR value; the dword index is uniform (SGPR).
+    OpdsMapping[0] = getVGPROpMapping(MI.getOperand(0).getReg(), MRI, *TRI);
+    OpdsMapping[1] = getSGPROpMapping(MI.getOperand(1).getReg(), MRI, *TRI);
+    break;
+  }
   case AMDGPU::G_AMDGPU_BUFFER_LOAD:
   case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE:
   case AMDGPU::G_AMDGPU_BUFFER_LOAD_SBYTE:
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSearchableTables.td b/llvm/lib/Target/AMDGPU/AMDGPUSearchableTables.td
index 81738d3fdc65a..b4a069efe1b3e 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSearchableTables.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSearchableTables.td
@@ -429,3 +429,20 @@ def AMDGPUImageDMaskIntrinsicTable : GenericTable {
   let PrimaryKeyName = "getAMDGPUImageDMaskIntrinsic";
   let PrimaryKeyEarlyOut = 1;
 }
+
+//===----------------------------------------------------------------------===//
+// V_LOAD/STORE_IDX opcode mapping table.
+//===----------------------------------------------------------------------===//
+
+def VLdStIdxOpcodeInfoTable : GenericTable {
+  let FilterClass = "VLdStIdxOpcodeInfo";
+  let CppTypeName = "VLdStIdxOpcodeInfo";
+  let Fields = ["Opcode", "BitWidth", "IsStore"];
+  let PrimaryKey = ["Opcode"];
+  let PrimaryKeyName = "getVLdStIdxOpcodeInfoByOpcodeImpl";
+}
+
+def getVLdStIdxOpcodeInfoByKeyImpl : SearchIndex {
+  let Table = VLdStIdxOpcodeInfoTable;
+  let Key = ["BitWidth", "IsStore"];
+}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
index 72c028edbaef5..a94541b304ee8 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
@@ -702,6 +702,7 @@ extern "C" LLVM_ABI LLVM_EXTERNAL_VISIBILITY void LLVMInitializeAMDGPUTarget() {
   initializeAMDGPURegBankLegalizeLegacyPass(*PR);
   initializeSILowerWWMCopiesLegacyPass(*PR);
   initializeAMDGPUMarkLastScratchLoadLegacyPass(*PR);
+  initializeAMDGPUAssignIdxToM0LegacyPass(*PR);
   initializeSILowerSGPRSpillsLegacyPass(*PR);
   initializeSIFixSGPRCopiesLegacyPass(*PR);
   initializeSIFixVGPRCopiesLegacyPass(*PR);
@@ -1846,6 +1847,10 @@ void GCNPassConfig::addFastRegAlloc() {
 }
 
 void GCNPassConfig::addPreRegAlloc() {
+  // Copy the VGPR "as memory" load/store index into M0 before register
+  // allocation; the movrel emitted later by AMDGPULowerVGPREncoding reads it.
+  addPass(&AMDGPUAssignIdxToM0ID);
+
   if (getOptLevel() != CodeGenOptLevel::None)
     addPass(&AMDGPUPrepareAGPRAllocLegacyID);
   if (getOptLevel() >= CodeGenOptLevel::Default && EnableMachinePipeliner)
@@ -2697,6 +2702,10 @@ Error AMDGPUCodeGenPassBuilder::addOptimizedRegAlloc(PassManagerWrapper &PMW) {
 }
 
 void AMDGPUCodeGenPassBuilder::addPreRegAlloc(PassManagerWrapper &PMW) {
+  // Set up M0 for the movrel that expands a VGPR "as memory" indexed access.
+  // Run before allocation so the index computation coalesces into M0.
+  addMachineFunctionPass(AMDGPUAssignIdxToM0Pass(), PMW);
+
   if (getOptLevel() != CodeGenOptLevel::None)
     addMachineFunctionPass(AMDGPUPrepareAGPRAllocPass(), PMW);
   if (getOptLevel() >= CodeGenOptLevel::Default && EnableMachinePipeliner)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index b32e3e1def7d3..a8ae3a15057f6 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1208,13 +1208,16 @@ bool GCNTTIImpl::isSourceOfDivergence(const Value *V) const {
 
   // Loads from the private and flat address spaces are divergent, because
   // threads can execute the load instruction with the same inputs and get
-  // different results.
+  // different results. The same is true of the VGPR ("as memory") address
+  // space: it is a per-lane view of the vector registers, so an access at a
+  // uniform offset still yields a per-lane (divergent) value.
   //
   // All other loads are not divergent, because if threads issue loads with the
   // same arguments, they will always get the same result.
   if (const LoadInst *Load = dyn_cast<LoadInst>(V))
     return Load->getPointerAddressSpace() == AMDGPUAS::PRIVATE_ADDRESS ||
-           Load->getPointerAddressSpace() == AMDGPUAS::FLAT_ADDRESS;
+           Load->getPointerAddressSpace() == AMDGPUAS::FLAT_ADDRESS ||
+           Load->getPointerAddressSpace() == AMDGPUAS::VGPR;
 
   // Atomics are divergent because they are executed sequentially: when an
   // atomic operation refers to the same address in each thread, then each
diff --git a/llvm/lib/Target/AMDGPU/CMakeLists.txt b/llvm/lib/Target/AMDGPU/CMakeLists.txt
index 4a5f77d55afa5..cc231e20b13f6 100644
--- a/llvm/lib/Target/AMDGPU/CMakeLists.txt
+++ b/llvm/lib/Target/AMDGPU/CMakeLists.txt
@@ -47,6 +47,7 @@ add_llvm_target(AMDGPUCodeGen
   AMDGPUArgumentUsageInfo.cpp
   AMDGPUAsanInstrumentation.cpp
   AMDGPUAsmPrinter.cpp
+  AMDGPUAssignIdxToM0.cpp
   AMDGPUAtomicOptimizer.cpp
   AMDGPUAttributor.cpp
   AMDGPUBarrierLatency.cpp
@@ -83,6 +84,7 @@ add_llvm_target(AMDGPUCodeGen
   AMDGPULowerExecSync.cpp
   AMDGPUSwLowerLDS.cpp
   AMDGPUMachineFunctionInfo.cpp
+  AMDGPUMachineInstrs.cpp
   AMDGPUMachineModuleInfo.cpp
   AMDGPUMacroFusion.cpp
   AMDGPUMCInstLower.cpp
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 40fcc2e25525a..2732c91115198 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -16,6 +16,7 @@
 #include "AMDGPUIGroupLP.h"
 #include "AMDGPUInstrInfo.h"
 #include "AMDGPULaneMaskUtils.h"
+#include "AMDGPUMachineInstrs.h"
 #include "AMDGPUMemoryUtils.h"
 #include "AMDGPUSelectionDAGInfo.h"
 #include "AMDGPUTargetMachine.h"
@@ -13597,6 +13598,87 @@ static bool addressMayBeAccessedAsPrivate(const MachineMemOperand *MMO,
   return true;
 }
 
+// Lower a load or store of the VGPR ("as memory") address space (13) to a
+// REG_LOAD / REG_STORE target node. The 32-bit pointer is a byte offset into
+// the wave's view of its vector registers; the target node carries the dword
+// index (pointer >> 2). Recognizing a constant dword offset is left to the
+// selection patterns, which fold an (add index, imm) shape into the pseudo.
+//
+// TODO: sub-dword (8/16-bit) accesses are not yet supported; they are
+// diagnosed as unsupported below.
+SDValue SITargetLowering::LowerLoadStoreVGPR(SDValue Op,
+                                             SelectionDAG &DAG) const {
+  SDLoc DL(Op);
+  MemSDNode *MemOp = cast<MemSDNode>(Op);
+  EVT MemVT = MemOp->getMemoryVT();
+  unsigned BitWidth = MemVT.getSizeInBits();
+
+  // Only whole-dword, non-extending/non-truncating accesses are implemented.
+  // Reject anything else with a diagnostic (replacing the value with poison)
+  // instead of failing instruction selection. Both callers - operation
+  // legalization and the pre-ISel combine - replace the node with this result,
+  // so the diagnostic is emitted exactly once.
+  auto reportUnsupported = [&]() -> SDValue {
+    const Function &F = DAG.getMachineFunction().getFunction();
+    DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
+        F,
+        "unsupported access of VGPR 'as memory' address space (13); only "
+        "whole-dword loads and stores are implemented",
+        DL.getDebugLoc()));
+    if (isa<StoreSDNode>(MemOp))
+      return MemOp->getChain();
+    return DAG.getMergeValues(
+        {DAG.getPOISON(Op.getValueType()), MemOp->getChain()}, DL);
+  };
+
+  if (BitWidth < 32)
+    return reportUnsupported();
+  if (auto *Load = dyn_cast<LoadSDNode>(MemOp)) {
+    if (Load->getExtensionType() != ISD::NON_EXTLOAD)
+      return reportUnsupported();
+    if (AMDGPUMI::VLoadIdxInst::tryGetOpcodeForBitWidth(BitWidth) == -1)
+      return reportUnsupported();
+  } else {
+    auto *Store = cast<StoreSDNode>(MemOp);
+    if (Store->isTruncatingStore())
+      return reportUnsupported();
+    if (AMDGPUMI::VStoreIdxInst::tryGetOpcodeForBitWidth(BitWidth) == -1)
+      return reportUnsupported();
+  }
+
+  SDValue Chain = MemOp->getChain();
+  SDValue Index = DAG.getNode(ISD::SRL, DL, MVT::i32, MemOp->getBasePtr(),
+                              DAG.getConstant(2, DL, MVT::i32));
+
+  // View the access as i32 / <N x i32> when the memory type is not register
+  // legal (e.g. v4i8), bitcasting the value across.
+  EVT RegVT = MemVT;
+  if (!isTypeLegal(RegVT)) {
+    unsigned NumDwords = BitWidth / 32;
+    RegVT = NumDwords == 1
+                ? EVT(MVT::i32)
+                : EVT::getVectorVT(*DAG.getContext(), MVT::i32, NumDwords);
+  }
+
+  if (auto *StoreOp = dyn_cast<StoreSDNode>(MemOp)) {
+    SDValue Value = StoreOp->getValue();
+    if (RegVT != MemVT)
+      Value = DAG.getNode(ISD::BITCAST, DL, RegVT, Value);
+    return DAG.getMemIntrinsicNode(
+        AMDGPUISD::REG_STORE, DL, DAG.getVTList(MVT::Other),
+        {Chain, Value, Index}, MemVT, StoreOp->getMemOperand());
+  }
+
+  auto *LoadOp = cast<LoadSDNode>(MemOp);
+  SDValue NewLoad = DAG.getMemIntrinsicNode(
+      AMDGPUISD::REG_LOAD, DL, DAG.getVTList(RegVT, MVT::Other), {Chain, Index},
+      MemVT, LoadOp->getMemOperand());
+  if (RegVT == MemVT)
+    return NewLoad;
+  SDValue Value = DAG.getNode(ISD::BITCAST, DL, MemVT, NewLoad);
+  return DAG.getMergeValues({Value, NewLoad.getValue(1)}, DL);
+}
+
 SDValue SITargetLowering::LowerLOAD(SDValue Op, SelectionDAG &DAG) const {
   SDLoc DL(Op);
   LoadSDNode *Load = cast<LoadSDNode>(Op);
@@ -13604,6 +13686,9 @@ SDValue SITargetLowering::LowerLOAD(SDValue Op, SelectionDAG &DAG) const {
   EVT MemVT = Load->getMemoryVT();
   MachineMemOperand *MMO = Load->getMemOperand();
 
+  if (Load->getAddressSpace() == AMDGPUAS::VGPR)
+    return LowerLoadStoreVGPR(Op, DAG);
+
   if (ExtType == ISD::NON_EXTLOAD && MemVT.getSizeInBits() < 32) {
     if (MemVT == MVT::i16 && isTypeLegal(MVT::i16))
       return SDValue();
@@ -14276,6 +14361,9 @@ SDValue SITargetLowering::LowerSTORE(SDValue Op, SelectionDAG &DAG) const {
   StoreSDNode *Store = cast<StoreSDNode>(Op);
   EVT VT = Store->getMemoryVT();
 
+  if (Store->getAddressSpace() == AMDGPUAS::VGPR)
+    return LowerLoadStoreVGPR(Op, DAG);
+
   if (VT == MVT::i1) {
     return DAG.getTruncStore(
         Store->getChain(), DL,
@@ -19272,6 +19360,19 @@ SDValue SITargetLowering::PerformDAGCombine(SDNode *N,
     if (auto Res = promoteUniformOpToI32(SDValue(N, 0), DCI))
       return Res;
     break;
+  case ISD::LOAD:
+    // Lower a VGPR ("as memory") address space (13) load to a REG_LOAD target
+    // node. Done here (not via operation legalization) so it also fires at -O0,
+    // where a scalar load is otherwise Legal and never reaches LowerLOAD.
+    if (cast<LoadSDNode>(N)->getAddressSpace() == AMDGPUAS::VGPR)
+      if (SDValue V = LowerLoadStoreVGPR(SDValue(N, 0), DCI.DAG))
+        return V;
+    break;
+  case ISD::STORE:
+    if (cast<StoreSDNode>(N)->getAddressSpace() == AMDGPUAS::VGPR)
+      if (SDValue V = LowerLoadStoreVGPR(SDValue(N, 0), DCI.DAG))
+        return V;
+    break;
   default:
     break;
   }
@@ -19873,6 +19974,22 @@ void SITargetLowering::AdjustInstrPostInstrSelection(MachineInstr &MI,
     return;
   }
 
+  // A VGPR "as memory" indexed load/store with a register index expands (on
+  // movrel subtargets) to an M0-relative move. Add an implicit-def of $m0: it
+  // records that the eventual move clobbers M0, and - because an instruction
+  // defining a physical register is not hoisted/sunk - keeps a divergent access
+  // pinned inside its waterfall loop. AMDGPUAssignIdxToM0 removes this when it
+  // writes M0 for real.
+  if (auto *LdStIdx = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
+    if (getSubtarget()->hasMovrel()) {
+      MachineOperand &IdxOp = LdStIdx->getIdxOp();
+      if (IdxOp.isReg())
+        MI.addOperand(MachineOperand::CreateReg(AMDGPU::M0, /*isDef=*/true,
+                                                /*isImp=*/true));
+    }
+    return;
+  }
+
   if (TII->isImage(MI))
     TII->enforceOperandRCAlignment(MI, AMDGPU::OpName::vaddr);
 }
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.h b/llvm/lib/Target/AMDGPU/SIISelLowering.h
index 2ce36586402b6..d3b9c708ff644 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.h
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.h
@@ -128,6 +128,9 @@ class SITargetLowering final : public AMDGPUTargetLowering {
 
   SDValue widenLoad(LoadSDNode *Ld, DAGCombinerInfo &DCI) const;
   SDValue LowerLOAD(SDValue Op, SelectionDAG &DAG) const;
+  // Lower a load/store of the VGPR ("as memory") address space (13) to a
+  // REG_LOAD/REG_STORE target node indexed by the pointer's dword offset.
+  SDValue LowerLoadStoreVGPR(SDValue Op, SelectionDAG &DAG) const;
   SDValue LowerSELECT(SDValue Op, SelectionDAG &DAG) const;
   SDValue lowerFastUnsafeFDIV(SDValue Op, SelectionDAG &DAG) const;
   SDValue lowerFastUnsafeFDIV64(SDValue Op, SelectionDAG &DAG) const;
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index 0346ebc254361..a16a6f3fb2d71 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -15,6 +15,7 @@
 #include "AMDGPU.h"
 #include "AMDGPUInstrInfo.h"
 #include "AMDGPULaneMaskUtils.h"
+#include "AMDGPUMachineInstrs.h"
 #include "GCNHazardRecognizer.h"
 #include "GCNSubtarget.h"
 #include "SIMachineFunctionInfo.h"
@@ -663,6 +664,18 @@ bool SIInstrInfo::getMemOperandsWithOffsetWidth(
     return true;
   }
 
+  if (auto *LdStIdx = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&LdSt)) {
+    BaseOp = &LdStIdx->getIdxOp();
+    OffsetOp = &LdStIdx->getOffsetOp();
+
+    BaseOps.push_back(BaseOp);
+    Offset = OffsetOp->getImm() * 4; // Offset has units of dwords.
+
+    // Get appropriate operand, and compute width accordingly.
+    Width = LocationSize::precise(LdStIdx->getBitWidth() / 8);
+    return true;
+  }
+
   return false;
 }
 
@@ -4298,10 +4311,20 @@ bool SIInstrInfo::areMemAccessesTriviallyDisjoint(const MachineInstr &MIa,
   if (MIa.hasOrderedMemoryRef() || MIb.hasOrderedMemoryRef())
     return false;
 
-  if (isLDSDMA(MIa) || isLDSDMA(MIb))
+  if (MIa.isBundle() || MIb.isBundle())
     return false;
 
-  if (MIa.isBundle() || MIb.isBundle())
+  // VGPR "as memory" indexed accesses only alias each other, and then only
+  // when their [idx+offset, idx+offset+width) dword ranges overlap.
+  const bool IsLdStIdxA = isa<AMDGPUMI::VLoadStoreIdxInst>(MIa);
+  const bool IsLdStIdxB = isa<AMDGPUMI::VLoadStoreIdxInst>(MIb);
+  if (IsLdStIdxA || IsLdStIdxB) {
+    if (IsLdStIdxA && IsLdStIdxB)
+      return checkInstOffsetsDoNotOverlap(MIa, MIb);
+    return true;
+  }
+
+  if (isLDSDMA(MIa) || isLDSDMA(MIb))
     return false;
 
   // TODO: Should we check the address space from the MachineMemOperand? That
@@ -5886,6 +5909,7 @@ bool SIInstrInfo::verifyInstruction(const MachineInstr &MI,
       Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64) {
     const bool IsDst = Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e32 ||
                        Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64;
+    const bool IsPreRA = !MI.getMF()->getProperties().hasNoVRegs();
 
     const unsigned StaticNumOps =
         Desc.getNumOperands() + Desc.implicit_uses().size();
@@ -5895,7 +5919,7 @@ bool SIInstrInfo::verifyInstruction(const MachineInstr &MI,
     // post RA scheduler where the main implicit operand is killed and
     // implicit-defs are added for sub-registers that remain live after this
     // instruction.
-    if (MI.getNumOperands() < StaticNumOps + NumImplicitOps) {
+    if (IsPreRA && MI.getNumOperands() < StaticNumOps + NumImplicitOps) {
       ErrInfo = "missing implicit register operands";
       return false;
     }
@@ -5908,20 +5932,22 @@ bool SIInstrInfo::verifyInstruction(const MachineInstr &MI,
       }
 
       unsigned UseOpIdx;
-      if (!MI.isRegTiedToUseOperand(StaticNumOps, &UseOpIdx) ||
-          UseOpIdx != StaticNumOps + 1) {
+      if (IsPreRA && (!MI.isRegTiedToUseOperand(StaticNumOps, &UseOpIdx) ||
+                      UseOpIdx != StaticNumOps + 1)) {
         ErrInfo = "movrel implicit operands should be tied";
         return false;
       }
     }
 
-    const MachineOperand &Src0 = MI.getOperand(Src0Idx);
-    const MachineOperand &ImpUse
-      = MI.getOperand(StaticNumOps + NumImplicitOps - 1);
-    if (!ImpUse.isReg() || !ImpUse.isUse() ||
-        !isSubRegOf(RI, ImpUse, IsDst ? *Dst : Src0)) {
-      ErrInfo = "src0 should be subreg of implicit vector use";
-      return false;
+    if (IsPreRA) {
+      const MachineOperand &Src0 = MI.getOperand(Src0Idx);
+      const MachineOperand &ImpUse =
+          MI.getOperand(StaticNumOps + NumImplicitOps - 1);
+      if (!ImpUse.isReg() || !ImpUse.isUse() ||
+          !isSubRegOf(RI, ImpUse, IsDst ? *Dst : Src0)) {
+        ErrInfo = "src0 should be subreg of implicit vector use";
+        return false;
+      }
     }
   }
 
@@ -7757,6 +7783,17 @@ SIInstrInfo::legalizeOperands(MachineInstr &MI,
     return CreatedBB;
   }
 
+  // A VGPR "as memory" indexed load/store needs its dword index in an SGPR (it
+  // becomes M0). A divergent (VGPR) index is made uniform with a waterfall
+  // loop that executes the access once per unique index across the wave.
+  if (auto *LdStIdx = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
+    MachineOperand *Idx = &LdStIdx->getIdxOp();
+    if (Idx->isReg() && Idx->getReg().isVirtual() &&
+        !RI.isSGPRClass(MRI.getRegClass(Idx->getReg())))
+      CreatedBB = generateWaterFallLoop(*this, MI, {Idx}, MDT);
+    return CreatedBB;
+  }
+
   // Legalize PHI
   // The register class of the operands must be the same type as the register
   // class of the output.
@@ -11274,9 +11311,16 @@ SIInstrInfo::getGenericValueUniformity(const MachineInstr &MI) const {
     return ValueUniformity::Default;
   }
 
+  // A VGPR ("as memory") indexed load is always divergent: it reads the wave's
+  // per-lane view of its vector registers, so even a uniform index yields a
+  // per-lane (divergent) value.
+  if (Opcode == AMDGPU::G_AMDGPU_REG_LOAD)
+    return ValueUniformity::NeverUniform;
+
   // Loads from the private and flat address spaces are divergent, because
   // threads can execute the load instruction with the same inputs and get
-  // different results.
+  // different results. The VGPR address space is likewise divergent (see
+  // above; this covers a G_LOAD not yet legalized to G_AMDGPU_REG_LOAD).
   //
   // All other loads are not divergent, because if threads issue loads with the
   // same arguments, they will always get the same result.
@@ -11287,7 +11331,8 @@ SIInstrInfo::getGenericValueUniformity(const MachineInstr &MI) const {
 
     if (llvm::any_of(MI.memoperands(), [](const MachineMemOperand *mmo) {
           return mmo->getAddrSpace() == AMDGPUAS::PRIVATE_ADDRESS ||
-                 mmo->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS;
+                 mmo->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS ||
+                 mmo->getAddrSpace() == AMDGPUAS::VGPR;
         })) {
       // At least one MMO in a non-global address space.
       return ValueUniformity::NeverUniform;
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.td b/llvm/lib/Target/AMDGPU/SIInstrInfo.td
index 508e8ebbdf155..3a7960f6abfb0 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.td
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.td
@@ -59,6 +59,17 @@ def GFX10Gen         : GFXGen<isGFX10Only, "GFX10", "_gfx10", SIEncodingFamily.G
 // modifier behavior with dx10_enable.
 def AMDGPUclamp : SDNode<"AMDGPUISD::CLAMP", SDTFPUnaryOp>;
 
+// VGPR address space (13) load/store with a dword index operand. The index is
+// the byte offset into the wave's view of its vector registers, divided by 4.
+def SDTRegIdxLoad : SDTypeProfile<1, 1,
+    [SDTCisVT<1, i32>]>; // dword_index
+def SDTRegIdxStore : SDTypeProfile<0, 2,
+    [SDTCisVT<1, i32>]>; // data, dword_index
+def SIreg_load : SDNode<"AMDGPUISD::REG_LOAD", SDTRegIdxLoad,
+                        [SDNPHasChain, SDNPMayLoad, SDNPMemOperand]>;
+def SIreg_store : SDNode<"AMDGPUISD::REG_STORE", SDTRegIdxStore,
+                         [SDNPHasChain, SDNPMayStore, SDNPMemOperand]>;
+
 def SDTSBufferLoad : SDTypeProfile<1, 3,
     [                    // vdata
      SDTCisVT<1, v4i32>, // rsrc
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index 3cf20ffb1fcf4..af474d2f297d0 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1060,6 +1060,88 @@ def SI_INDIRECT_DST_V32 : SI_INDIRECT_DST<VReg_1024>;
 
 } // End Uses = [EXEC], Defs = [M0, EXEC]
 
+//===----------------------------------------------------------------------===//
+// VGPR "as memory" indexed load/store pseudos (address space 13)
+//===----------------------------------------------------------------------===//
+
+// V_LOAD_IDX_B<N> / V_STORE_IDX_B<N> load or store N bits from/to the wave's
+// view of its vector registers, at a dword index of ($idx + $offset). $idx is
+// a 32-bit value that may be uniform (SGPR) or divergent (VGPR); $offset is a
+// constant dword offset folded in at selection.
+//
+// AMDGPUAssignIdxToM0 copies $idx into M0 before register allocation, then
+// AMDGPULowerVGPREncoding lowers each into v_movrels_b32 (load) /
+// v_movreld_b32 (store) over the wave's vector registers.
+
+// Populates the VLdStIdxOpcodeInfo searchable table (see AMDGPUMachineInstrs.h
+// and AMDGPU::getVLdStIdxOpcodeInfo*), mapping each pseudo to its bit width and
+// load/store direction.
+class VLdStIdxOpcodeInfo<int size, bit isStore> {
+  Instruction Opcode = !cast<Instruction>(NAME);
+  bits<12> BitWidth = size;
+  bit IsStore = isStore;
+}
+
+foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
+              VReg_224, VReg_256, VReg_288, VReg_320, VReg_352, VReg_384,
+              VReg_512, VReg_1024] in {
+  // The units of $idx and $offset are in dwords.
+  //
+  // VALU adds an implicit $exec use; together with the implicit-def $m0 added
+  // by AdjustInstrPostInstrSelection (see hasPostISelHook), this keeps a
+  // divergent-index access pinned inside its waterfall loop rather than being
+  // hoisted/sunk out.
+  def V_LOAD_IDX_B#rc.Size : VPseudoInstSI <
+    (outs rc:$data),
+    (ins SReg_32:$idx, i32imm:$offset)>,
+    VLdStIdxOpcodeInfo<rc.Size, 0> {
+      let mayLoad = 1;
+      let VALU = 1;
+      let UseNamedOperandTable = 1;
+      let hasSideEffects = 0;
+      let hasPostISelHook = 1;
+  }
+  def V_STORE_IDX_B#rc.Size : VPseudoInstSI <
+    (outs),
+    (ins rc:$data, SReg_32:$idx, i32imm:$offset)>,
+    VLdStIdxOpcodeInfo<rc.Size, 1> {
+      let mayStore = 1;
+      let VALU = 1;
+      let UseNamedOperandTable = 1;
+      let hasSideEffects = 0;
+      let hasPostISelHook = 1;
+  }
+}
+
+// Select the REG_LOAD/REG_STORE target nodes into the sized indexed pseudos.
+// The pointer is a byte offset into the file; SIreg_load/SIreg_store carry the
+// dword index (ptr >> 2), and an (add idx, imm) shape folds a constant dword
+// offset into the pseudo's $offset operand.
+multiclass VRegIdxLoadStorePat<ValueType vt> {
+  defvar load_inst = !cast<Instruction>("V_LOAD_IDX_B"#vt.Size);
+  defvar store_inst = !cast<Instruction>("V_STORE_IDX_B"#vt.Size);
+
+  def : GCNPat<
+    (vt (SIreg_load (add i32:$idx, (i32 imm:$offset)))),
+    (load_inst $idx, imm:$offset)>;
+  def : GCNPat<
+    (vt (SIreg_load i32:$idx)),
+    (load_inst $idx, 0)>;
+  def : GCNPat<
+    (SIreg_store vt:$data, (add i32:$idx, (i32 imm:$offset))),
+    (store_inst $data, $idx, imm:$offset)>;
+  def : GCNPat<
+    (SIreg_store vt:$data, i32:$idx),
+    (store_inst $data, $idx, 0)>;
+}
+
+foreach vt = !listconcat(
+    Reg32Types.types, Reg64Types.types, Reg96Types.types, Reg128Types.types,
+    Reg160Types.types, Reg192Types.types, Reg224Types.types, Reg256Types.types,
+    Reg288Types.types, Reg320Types.types, Reg352Types.types, Reg384Types.types,
+    Reg512Types.types, Reg1024Types.types) in
+defm : VRegIdxLoadStorePat<vt>;
+
 // This is a pseudo variant of the v_movreld_b32 instruction in which the
 // vector operand appears only twice, once as def and once as use. Using this
 // pseudo avoids problems with the Two Address instructions pass.
@@ -4827,6 +4909,24 @@ def G_AMDGPU_BUFFER_STORE_FORMAT_D16 : BufferStoreGenericInstruction;
 def G_AMDGPU_TBUFFER_STORE_FORMAT : TBufferStoreGenericInstruction;
 def G_AMDGPU_TBUFFER_STORE_FORMAT_D16 : TBufferStoreGenericInstruction;
 
+// GlobalISel equivalents of the REG_LOAD / REG_STORE target nodes: a load or
+// store of the VGPR ("as memory") address space, indexed by the pointer's
+// dword offset. They select to the V_LOAD_IDX_B<N> / V_STORE_IDX_B<N> pseudos
+// through the same TableGen patterns (see GINodeEquiv in AMDGPUGISel.td).
+def G_AMDGPU_REG_LOAD : AMDGPUGenericInstruction {
+  let OutOperandList = (outs type0:$dst);
+  let InOperandList = (ins type1:$dword_index);
+  let hasSideEffects = 0;
+  let mayLoad = 1;
+}
+
+def G_AMDGPU_REG_STORE : AMDGPUGenericInstruction {
+  let OutOperandList = (outs);
+  let InOperandList = (ins type0:$data, type1:$dword_index);
+  let hasSideEffects = 0;
+  let mayStore = 1;
+}
+
 def G_AMDGPU_FMIN_LEGACY : AMDGPUGenericInstruction {
   let OutOperandList = (outs type0:$dst);
   let InOperandList = (ins type0:$src0, type0:$src1);
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index e45359719d6c9..1f9faa076c978 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -480,9 +480,20 @@ struct FP4FP8DstByteSelInfo {
 #define GET_getMFMA_F8F6F4_WithSize_IMPL
 #define GET_isMFMA_F8F6F4Table_IMPL
 #define GET_isCvtScaleF32_F32F16ToF8F4Table_IMPL
+#define GET_VLdStIdxOpcodeInfoTable_DECL
+#define GET_VLdStIdxOpcodeInfoTable_IMPL
 
 #include "AMDGPUGenSearchableTables.inc"
 
+const VLdStIdxOpcodeInfo *getVLdStIdxOpcodeInfoByOpcode(unsigned Opc) {
+  return getVLdStIdxOpcodeInfoByOpcodeImpl(Opc);
+}
+
+const VLdStIdxOpcodeInfo *getVLdStIdxOpcodeInfoByKey(uint16_t BitWidth,
+                                                     bool IsStore) {
+  return getVLdStIdxOpcodeInfoByKeyImpl(BitWidth, IsStore);
+}
+
 int getMTBUFBaseOpcode(unsigned Opc) {
   const MTBUFInfo *Info = getMTBUFInfoFromOpcode(Opc);
   return Info ? Info->BaseOpcode : -1;
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index bce059a0c18a7..d6a6bccef5f59 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -485,6 +485,19 @@ struct MIMGInfo {
 LLVM_READONLY
 const MIMGInfo *getMIMGInfo(unsigned Opc);
 
+struct VLdStIdxOpcodeInfo {
+  unsigned Opcode;
+  uint16_t BitWidth;
+  bool IsStore;
+};
+
+LLVM_READONLY
+const VLdStIdxOpcodeInfo *getVLdStIdxOpcodeInfoByOpcode(unsigned Opc);
+
+LLVM_READONLY
+const VLdStIdxOpcodeInfo *getVLdStIdxOpcodeInfoByKey(uint16_t BitWidth,
+                                                     bool IsStore);
+
 LLVM_READONLY
 int getMTBUFBaseOpcode(unsigned Opc);
 
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
new file mode 100644
index 0000000000000..08d3095de3561
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
@@ -0,0 +1,384 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+
+; End-to-end lowering of the VGPR "as memory" address space (13) on a
+; movrel-capable subtarget (gfx12). A load/store of a uniform (SGPR) pointer
+; lowers to an M0-relative move (v_movrels_b32 / v_movreld_b32) over the wave's
+; vector registers, with the dword index (pointer >> 2) placed in M0.
+
+define i32 @load_i32(ptr addrspace(13) inreg %p) {
+; GFX12-LABEL: load_i32:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT:    v_add_nc_u32_e32 v0, 1, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %x = load i32, ptr addrspace(13) %p
+  %y = add i32 %x, 1
+  ret i32 %y
+}
+
+define i64 @load_i64(ptr addrspace(13) inreg %p) {
+; GFX12-LABEL: load_i64:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-NEXT:    v_add_co_u32 v0, vcc_lo, v0, 1
+; GFX12-NEXT:    s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT:    v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %x = load i64, ptr addrspace(13) %p
+  %y = add i64 %x, 1
+  ret i64 %y
+}
+
+define <2 x float> @load_v2f32(ptr addrspace(13) inreg %p) {
+; GFX12-LABEL: load_v2f32:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT:    v_dual_add_f32 v0, 0x42280000, v0 :: v_dual_add_f32 v1, 0x42280000, v1
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <2 x float>, ptr addrspace(13) %p
+  %y = fadd <2 x float> %x, <float 42.0, float 42.0>
+  ret <2 x float> %y
+}
+
+define <3 x float> @load_v3f32(ptr addrspace(13) inreg %p) {
+; GFX12-SDAG-LABEL: load_v3f32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-SDAG-NEXT:    s_add_co_i32 s0, s0, 64
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v3, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v4, v1
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v5, v2
+; GFX12-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-SDAG-NEXT:    v_dual_add_f32 v0, v0, v3 :: v_dual_add_f32 v1, v1, v4
+; GFX12-SDAG-NEXT:    v_add_f32_e32 v2, v2, v5
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: load_v3f32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-GISEL-NEXT:    s_add_co_u32 s0, s0, 64
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v3, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v4, v1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v5, v2
+; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-GISEL-NEXT:    v_dual_add_f32 v0, v0, v3 :: v_dual_add_f32 v1, v1, v4
+; GFX12-GISEL-NEXT:    v_add_f32_e32 v2, v2, v5
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %p.2 = getelementptr i32, ptr addrspace(13) %p, i32 16
+  %x = load <3 x float>, ptr addrspace(13) %p
+  %y = load <3 x float>, ptr addrspace(13) %p.2
+  %z = fadd <3 x float> %x, %y
+  ret <3 x float> %z
+}
+
+define void @store_i32(ptr addrspace(13) inreg %p, i32 %x) {
+; GFX12-LABEL: store_i32:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    v_add_nc_u32_e32 v0, 1, v0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %y = add i32 %x, 1
+  store i32 %y, ptr addrspace(13) %p
+  ret void
+}
+
+define void @store_i64(ptr addrspace(13) inreg %p, i64 %x, i64 %y) {
+; GFX12-LABEL: store_i64:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    v_add_co_u32 v0, vcc_lo, v0, v2
+; GFX12-NEXT:    s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT:    v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    v_movreld_b32_e32 v1, v1
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %z = add i64 %x, %y
+  store i64 %z, ptr addrspace(13) %p
+  ret void
+}
+
+define void @store_v3i32(ptr addrspace(13) inreg %p, <3 x i32> %x, <3 x i32> %y) {
+; GFX12-SDAG-LABEL: store_v3i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    v_add_nc_u32_e32 v2, v2, v5
+; GFX12-SDAG-NEXT:    v_add_nc_u32_e32 v1, v1, v4
+; GFX12-SDAG-NEXT:    v_add_nc_u32_e32 v0, v0, v3
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX12-SDAG-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movreld_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movreld_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: store_v3i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    v_add_nc_u32_e32 v0, v0, v3
+; GFX12-GISEL-NEXT:    v_add_nc_u32_e32 v1, v1, v4
+; GFX12-GISEL-NEXT:    v_add_nc_u32_e32 v2, v2, v5
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX12-GISEL-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movreld_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_3)
+; GFX12-GISEL-NEXT:    v_movreld_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %z = add <3 x i32> %x, %y
+  store <3 x i32> %z, ptr addrspace(13) %p
+  ret void
+}
+
+define void @store_v8f16(ptr addrspace(13) inreg %p, <8 x half> %x, <8 x half> %y) {
+; GFX12-SDAG-LABEL: store_v8f16:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    v_pk_add_f16 v3, v3, v7
+; GFX12-SDAG-NEXT:    v_pk_add_f16 v2, v2, v6
+; GFX12-SDAG-NEXT:    v_pk_add_f16 v1, v1, v5
+; GFX12-SDAG-NEXT:    v_pk_add_f16 v0, v0, v4
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX12-SDAG-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movreld_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movreld_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    v_movreld_b32_e32 v3, v3
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: store_v8f16:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    v_pk_add_f16 v0, v0, v4
+; GFX12-GISEL-NEXT:    v_pk_add_f16 v1, v1, v5
+; GFX12-GISEL-NEXT:    v_pk_add_f16 v2, v2, v6
+; GFX12-GISEL-NEXT:    v_pk_add_f16 v3, v3, v7
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-GISEL-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movreld_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    v_movreld_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    v_movreld_b32_e32 v3, v3
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %z = fadd <8 x half> %x, %y
+  store <8 x half> %z, ptr addrspace(13) %p
+  ret void
+}
+
+define void @copy_i32(ptr addrspace(13) inreg %dst, ptr addrspace(13) inreg %src) {
+; GFX12-LABEL: copy_i32:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s1, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %x = load i32, ptr addrspace(13) %src
+  store i32 %x, ptr addrspace(13) %dst
+  ret void
+}
+
+define void @copy_v2i32_unaligned(ptr addrspace(13) inreg %dst, ptr addrspace(13) inreg %src) {
+; GFX12-LABEL: copy_v2i32_unaligned:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s1, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    v_movreld_b32_e32 v1, v1
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <2 x i32>, ptr addrspace(13) %src, align 4
+  store <2 x i32> %x, ptr addrspace(13) %dst, align 4
+  ret void
+}
+
+define void @copy_i64_aligned(ptr addrspace(13) inreg %dst, ptr addrspace(13) inreg %src) {
+; GFX12-LABEL: copy_i64_aligned:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s1, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    v_movreld_b32_e32 v1, v1
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %x = load i64, ptr addrspace(13) %src, align 8
+  store i64 %x, ptr addrspace(13) %dst, align 8
+  ret void
+}
+
+; Null and poison pointers must be accepted (produce valid code) rather than
+; crash or fail the machine verifier. The specific null-pointer value is
+; defined by the parent change that introduces the address space.
+
+define i32 @load_null() {
+; GFX12-LABEL: load_null:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_mov_b32 m0, 0
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %x = load i32, ptr addrspace(13) null
+  ret i32 %x
+}
+
+define void @store_null(i32 %v) {
+; GFX12-LABEL: store_null:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_mov_b32 m0, 0
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  store i32 %v, ptr addrspace(13) null
+  ret void
+}
+
+define i32 @load_poison() {
+; GFX12-SDAG-LABEL: load_poison:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_mov_b32 m0, 0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: load_poison:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load i32, ptr addrspace(13) poison
+  ret i32 %x
+}
+
+define void @store_poison(i32 %v) {
+; GFX12-SDAG-LABEL: store_poison:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_mov_b32 m0, 0
+; GFX12-SDAG-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: store_poison:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-GISEL-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  store i32 %v, ptr addrspace(13) poison
+  ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll
new file mode 100644
index 0000000000000..48d90ecaf2eff
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll
@@ -0,0 +1,461 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+
+; Copy from VGPR "as memory" (address space 13) to global memory across the
+; range of legal whole-dword access sizes (32 up to 1024 bits). Each uniform
+; load lowers to a sequence of per-dword M0-relative moves (v_movrels_b32).
+
+define void @copy_i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-SDAG-LABEL: copy_i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v0, 0
+; GFX12-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v0
+; GFX12-SDAG-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: copy_i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v1, 0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    global_store_b32 v1, v0, s[0:1]
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load i32, ptr addrspace(13) %in
+  store i32 %x, ptr addrspace(1) %out
+  ret void
+}
+
+define void @copy_v2i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-LABEL: copy_v2i32:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-NEXT:    v_mov_b32_e32 v2, 0
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <2 x i32>, ptr addrspace(13) %in
+  store <2 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
+
+define void @copy_v3i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-LABEL: copy_v3i32:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-NEXT:    v_mov_b32_e32 v3, 0
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-NEXT:    global_store_b96 v3, v[0:2], s[0:1]
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <3 x i32>, ptr addrspace(13) %in
+  store <3 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
+
+define void @copy_v4i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-LABEL: copy_v4i32:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-NEXT:    v_mov_b32_e32 v4, 0
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-NEXT:    global_store_b128 v4, v[0:3], s[0:1]
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <4 x i32>, ptr addrspace(13) %in
+  store <4 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
+
+define void @copy_v5i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-SDAG-LABEL: copy_v5i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v5, 0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-SDAG-NEXT:    s_clause 0x1
+; GFX12-SDAG-NEXT:    global_store_b32 v5, v4, s[0:1] offset:16
+; GFX12-SDAG-NEXT:    global_store_b128 v5, v[0:3], s[0:1]
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: copy_v5i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v5, 0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-GISEL-NEXT:    s_clause 0x1
+; GFX12-GISEL-NEXT:    global_store_b128 v5, v[0:3], s[0:1]
+; GFX12-GISEL-NEXT:    global_store_b32 v5, v4, s[0:1] offset:16
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <5 x i32>, ptr addrspace(13) %in
+  store <5 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
+
+define void @copy_v6i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-SDAG-LABEL: copy_v6i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v6, 0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-SDAG-NEXT:    s_clause 0x1
+; GFX12-SDAG-NEXT:    global_store_b64 v6, v[4:5], s[0:1] offset:16
+; GFX12-SDAG-NEXT:    global_store_b128 v6, v[0:3], s[0:1]
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: copy_v6i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v6, 0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-GISEL-NEXT:    s_clause 0x1
+; GFX12-GISEL-NEXT:    global_store_b128 v6, v[0:3], s[0:1]
+; GFX12-GISEL-NEXT:    global_store_b64 v6, v[4:5], s[0:1] offset:16
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <6 x i32>, ptr addrspace(13) %in
+  store <6 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
+
+define void @copy_v7i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-SDAG-LABEL: copy_v7i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v7, 0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-SDAG-NEXT:    s_clause 0x1
+; GFX12-SDAG-NEXT:    global_store_b96 v7, v[4:6], s[0:1] offset:16
+; GFX12-SDAG-NEXT:    global_store_b128 v7, v[0:3], s[0:1]
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: copy_v7i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v7, 0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-GISEL-NEXT:    s_clause 0x1
+; GFX12-GISEL-NEXT:    global_store_b128 v7, v[0:3], s[0:1]
+; GFX12-GISEL-NEXT:    global_store_b96 v7, v[4:6], s[0:1] offset:16
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <7 x i32>, ptr addrspace(13) %in
+  store <7 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
+
+define void @copy_v8i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-SDAG-LABEL: copy_v8i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v8, 0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-SDAG-NEXT:    s_clause 0x1
+; GFX12-SDAG-NEXT:    global_store_b128 v8, v[4:7], s[0:1] offset:16
+; GFX12-SDAG-NEXT:    global_store_b128 v8, v[0:3], s[0:1]
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: copy_v8i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v8, 0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-GISEL-NEXT:    s_clause 0x1
+; GFX12-GISEL-NEXT:    global_store_b128 v8, v[0:3], s[0:1]
+; GFX12-GISEL-NEXT:    global_store_b128 v8, v[4:7], s[0:1] offset:16
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <8 x i32>, ptr addrspace(13) %in
+  store <8 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
+
+define void @copy_v16i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-SDAG-LABEL: copy_v16i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v16, 0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v8, v8
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v9, v9
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v10, v10
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v11, v11
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v12, v12
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v13, v13
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v14, v14
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v15, v15
+; GFX12-SDAG-NEXT:    s_clause 0x3
+; GFX12-SDAG-NEXT:    global_store_b128 v16, v[12:15], s[0:1] offset:48
+; GFX12-SDAG-NEXT:    global_store_b128 v16, v[8:11], s[0:1] offset:32
+; GFX12-SDAG-NEXT:    global_store_b128 v16, v[4:7], s[0:1] offset:16
+; GFX12-SDAG-NEXT:    global_store_b128 v16, v[0:3], s[0:1]
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: copy_v16i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v16, 0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v8, v8
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v9, v9
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v10, v10
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v11, v11
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v12, v12
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v13, v13
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v14, v14
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v15, v15
+; GFX12-GISEL-NEXT:    s_clause 0x3
+; GFX12-GISEL-NEXT:    global_store_b128 v16, v[0:3], s[0:1]
+; GFX12-GISEL-NEXT:    global_store_b128 v16, v[4:7], s[0:1] offset:16
+; GFX12-GISEL-NEXT:    global_store_b128 v16, v[8:11], s[0:1] offset:32
+; GFX12-GISEL-NEXT:    global_store_b128 v16, v[12:15], s[0:1] offset:48
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <16 x i32>, ptr addrspace(13) %in
+  store <16 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
+
+define void @copy_v32i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-SDAG-LABEL: copy_v32i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v32, 0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v8, v8
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v9, v9
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v10, v10
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v11, v11
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v12, v12
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v13, v13
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v14, v14
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v15, v15
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v16, v16
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v17, v17
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v18, v18
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v19, v19
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v20, v20
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v21, v21
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v22, v22
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v23, v23
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v24, v24
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v25, v25
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v26, v26
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v27, v27
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v28, v28
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v29, v29
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v30, v30
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v31, v31
+; GFX12-SDAG-NEXT:    s_clause 0x7
+; GFX12-SDAG-NEXT:    global_store_b128 v32, v[28:31], s[0:1] offset:112
+; GFX12-SDAG-NEXT:    global_store_b128 v32, v[24:27], s[0:1] offset:96
+; GFX12-SDAG-NEXT:    global_store_b128 v32, v[20:23], s[0:1] offset:80
+; GFX12-SDAG-NEXT:    global_store_b128 v32, v[16:19], s[0:1] offset:64
+; GFX12-SDAG-NEXT:    global_store_b128 v32, v[12:15], s[0:1] offset:48
+; GFX12-SDAG-NEXT:    global_store_b128 v32, v[8:11], s[0:1] offset:32
+; GFX12-SDAG-NEXT:    global_store_b128 v32, v[4:7], s[0:1] offset:16
+; GFX12-SDAG-NEXT:    global_store_b128 v32, v[0:3], s[0:1]
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: copy_v32i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v32, 0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v8, v8
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v9, v9
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v10, v10
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v11, v11
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v12, v12
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v13, v13
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v14, v14
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v15, v15
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v16, v16
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v17, v17
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v18, v18
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v19, v19
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v20, v20
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v21, v21
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v22, v22
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v23, v23
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v24, v24
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v25, v25
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v26, v26
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v27, v27
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v28, v28
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v29, v29
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v30, v30
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v31, v31
+; GFX12-GISEL-NEXT:    s_clause 0x7
+; GFX12-GISEL-NEXT:    global_store_b128 v32, v[0:3], s[0:1]
+; GFX12-GISEL-NEXT:    global_store_b128 v32, v[4:7], s[0:1] offset:16
+; GFX12-GISEL-NEXT:    global_store_b128 v32, v[8:11], s[0:1] offset:32
+; GFX12-GISEL-NEXT:    global_store_b128 v32, v[12:15], s[0:1] offset:48
+; GFX12-GISEL-NEXT:    global_store_b128 v32, v[16:19], s[0:1] offset:64
+; GFX12-GISEL-NEXT:    global_store_b128 v32, v[20:23], s[0:1] offset:80
+; GFX12-GISEL-NEXT:    global_store_b128 v32, v[24:27], s[0:1] offset:96
+; GFX12-GISEL-NEXT:    global_store_b128 v32, v[28:31], s[0:1] offset:112
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <32 x i32>, ptr addrspace(13) %in
+  store <32 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
new file mode 100644
index 0000000000000..28aa0ff7f0fb9
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
@@ -0,0 +1,130 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+
+; A divergent (per-lane) dword index into VGPR "as memory" (address space 13)
+; is handled with a waterfall loop: for each unique index across the wave, set
+; M0 and do the M0-relative move under a matching-lane EXEC subset. The pointer
+; arrives in a VGPR (no inreg), so the index (pointer >> 2) is divergent.
+
+define i32 @load_i32(ptr addrspace(13) %p) {
+; GFX12-SDAG-LABEL: load_i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    v_lshrrev_b32_e32 v0, 2, v0
+; GFX12-SDAG-NEXT:    s_mov_b32 s0, exec_lo
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    s_mov_b32 s1, s0
+; GFX12-SDAG-NEXT:  .LBB0_1: ; =>This Inner Loop Header: Depth=1
+; GFX12-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX12-SDAG-NEXT:    v_readfirstlane_b32 s2, v0
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_va_sdst(0)
+; GFX12-SDAG-NEXT:    v_cmpx_eq_u32_e32 s2, v0
+; GFX12-SDAG-NEXT:    s_mov_b32 m0, s2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v0
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    s_and_not1_wrexec_b32 s1, s1
+; GFX12-SDAG-NEXT:    ; implicit-def: $vgpr0
+; GFX12-SDAG-NEXT:    s_cbranch_execnz .LBB0_1
+; GFX12-SDAG-NEXT:  ; %bb.2:
+; GFX12-SDAG-NEXT:    s_mov_b32 exec_lo, s0
+; GFX12-SDAG-NEXT:    v_add_nc_u32_e32 v0, 1, v1
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: load_i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    v_lshrrev_b32_e32 v0, 2, v0
+; GFX12-GISEL-NEXT:    s_mov_b32 s0, exec_lo
+; GFX12-GISEL-NEXT:  .LBB0_1: ; =>This Inner Loop Header: Depth=1
+; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-GISEL-NEXT:    v_readfirstlane_b32 s2, v0
+; GFX12-GISEL-NEXT:    s_mov_b32 s1, exec_lo
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_va_sdst(0)
+; GFX12-GISEL-NEXT:    v_cmpx_eq_u32_e32 s2, v0
+; GFX12-GISEL-NEXT:    s_mov_b32 m0, s2
+; GFX12-GISEL-NEXT:    ; implicit-def: $vgpr0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v0
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    s_xor_b32 exec_lo, exec_lo, s1
+; GFX12-GISEL-NEXT:    s_cbranch_execnz .LBB0_1
+; GFX12-GISEL-NEXT:  ; %bb.2:
+; GFX12-GISEL-NEXT:    s_mov_b32 exec_lo, s0
+; GFX12-GISEL-NEXT:    v_add_nc_u32_e32 v0, 1, v1
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load i32, ptr addrspace(13) %p
+  %y = add i32 %x, 1
+  ret i32 %y
+}
+
+define void @store_i32(ptr addrspace(13) %p, i32 %x) {
+; GFX12-SDAG-LABEL: store_i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    v_add_nc_u32_e32 v1, 1, v1
+; GFX12-SDAG-NEXT:    v_lshrrev_b32_e32 v0, 2, v0
+; GFX12-SDAG-NEXT:    s_mov_b32 s0, exec_lo
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    s_mov_b32 s1, s0
+; GFX12-SDAG-NEXT:  .LBB1_1: ; =>This Inner Loop Header: Depth=1
+; GFX12-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX12-SDAG-NEXT:    v_readfirstlane_b32 s2, v0
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_va_sdst(0)
+; GFX12-SDAG-NEXT:    v_cmpx_eq_u32_e32 s2, v0
+; GFX12-SDAG-NEXT:    s_mov_b32 m0, s2
+; GFX12-SDAG-NEXT:    v_movreld_b32_e32 v0, v1
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    s_and_not1_wrexec_b32 s1, s1
+; GFX12-SDAG-NEXT:    ; implicit-def: $vgpr0
+; GFX12-SDAG-NEXT:    ; implicit-def: $vgpr1
+; GFX12-SDAG-NEXT:    s_cbranch_execnz .LBB1_1
+; GFX12-SDAG-NEXT:  ; %bb.2:
+; GFX12-SDAG-NEXT:    s_mov_b32 exec_lo, s0
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: store_i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    v_add_nc_u32_e32 v1, 1, v1
+; GFX12-GISEL-NEXT:    v_lshrrev_b32_e32 v0, 2, v0
+; GFX12-GISEL-NEXT:    s_mov_b32 s0, exec_lo
+; GFX12-GISEL-NEXT:  .LBB1_1: ; =>This Inner Loop Header: Depth=1
+; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-GISEL-NEXT:    v_readfirstlane_b32 s2, v0
+; GFX12-GISEL-NEXT:    s_mov_b32 s1, exec_lo
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_va_sdst(0)
+; GFX12-GISEL-NEXT:    v_cmpx_eq_u32_e32 s2, v0
+; GFX12-GISEL-NEXT:    s_mov_b32 m0, s2
+; GFX12-GISEL-NEXT:    ; implicit-def: $vgpr0
+; GFX12-GISEL-NEXT:    v_movreld_b32_e32 v0, v1
+; GFX12-GISEL-NEXT:    ; implicit-def: $vgpr1
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    s_xor_b32 exec_lo, exec_lo, s1
+; GFX12-GISEL-NEXT:    s_cbranch_execnz .LBB1_1
+; GFX12-GISEL-NEXT:  ; %bb.2:
+; GFX12-GISEL-NEXT:    s_mov_b32 exec_lo, s0
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %y = add i32 %x, 1
+  store i32 %y, ptr addrspace(13) %p
+  ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX12: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-inttoptr.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-inttoptr.ll
new file mode 100644
index 0000000000000..393928d359f96
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-inttoptr.ll
@@ -0,0 +1,48 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+
+; A constant VGPR "as memory" (address space 13) pointer formed with inttoptr:
+; the dword index (256 >> 2 = 64) is a compile-time constant, so M0 is set from
+; that constant and the access is a single M0-relative move.
+
+define i32 @load_i32() {
+; GFX12-LABEL: load_i32:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_mov_b32 m0, 64
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT:    v_add_nc_u32_e32 v0, 1, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %p = inttoptr i32 256 to ptr addrspace(13)
+  %x = load i32, ptr addrspace(13) %p
+  %y = add i32 %x, 1
+  ret i32 %y
+}
+
+define void @store_i32(i32 %x) {
+; GFX12-LABEL: store_i32:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    v_add_nc_u32_e32 v0, 1, v0
+; GFX12-NEXT:    s_mov_b32 m0, 64
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %p = inttoptr i32 256 to ptr addrspace(13)
+  %y = add i32 %x, 1
+  store i32 %y, ptr addrspace(13) %p
+  ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX12-GISEL: {{.*}}
+; GFX12-SDAG: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
new file mode 100644
index 0000000000000..40263e1c975ad
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
@@ -0,0 +1,31 @@
+; RUN: not llc -global-isel=0 -mtriple=amdgcn -mcpu=gfx1200 -filetype=null %s 2>&1 | FileCheck %s
+; RUN: not llc -global-isel=1 -mtriple=amdgcn -mcpu=gfx1200 -filetype=null %s 2>&1 | FileCheck %s
+
+; Sub-dword (8/16-bit) accesses of the VGPR "as memory" address space (13) are
+; not yet implemented. They must be rejected with a clean diagnostic on both
+; SelectionDAG and GlobalISel, rather than failing with "cannot select" /
+; "unable to legalize".
+
+; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only whole-dword loads and stores are implemented
+define i8 @load_i8(ptr addrspace(13) inreg %p) {
+  %x = load i8, ptr addrspace(13) %p
+  ret i8 %x
+}
+
+; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only whole-dword loads and stores are implemented
+define i16 @load_i16(ptr addrspace(13) inreg %p) {
+  %x = load i16, ptr addrspace(13) %p
+  ret i16 %x
+}
+
+; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only whole-dword loads and stores are implemented
+define void @store_i8(ptr addrspace(13) inreg %p, i8 %v) {
+  store i8 %v, ptr addrspace(13) %p
+  ret void
+}
+
+; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only whole-dword loads and stores are implemented
+define void @store_i16(ptr addrspace(13) inreg %p, i16 %v) {
+  store i16 %v, ptr addrspace(13) %p
+  ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll b/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll
index e346eb0a3b072..3f3e2b3a314a9 100644
--- a/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll
+++ b/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll
@@ -68,6 +68,7 @@
 ; GCN-O0-NEXT:   function
 ; GCN-O0-NEXT:     machine-function
 ; GCN-O0-NEXT:       reg-usage-propagation
+; GCN-O0-NEXT:       amdgpu-assign-idx-to-m0
 ; GCN-O0-NEXT:       phi-node-elimination
 ; GCN-O0-NEXT:       si-lower-control-flow
 ; GCN-O0-NEXT:       two-address-instruction
@@ -215,6 +216,7 @@
 ; GCN-O2-NEXT:   function
 ; GCN-O2-NEXT:     machine-function
 ; GCN-O2-NEXT:       reg-usage-propagation
+; GCN-O2-NEXT:       amdgpu-assign-idx-to-m0
 ; GCN-O2-NEXT:       amdgpu-prepare-agpr-alloc
 ; GCN-O2-NEXT:       detect-dead-lanes
 ; GCN-O2-NEXT:       dead-mi-elimination
@@ -403,6 +405,7 @@
 ; GCN-O3-NEXT:   function
 ; GCN-O3-NEXT:     machine-function
 ; GCN-O3-NEXT:       reg-usage-propagation
+; GCN-O3-NEXT:       amdgpu-assign-idx-to-m0
 ; GCN-O3-NEXT:       amdgpu-prepare-agpr-alloc
 ; GCN-O3-NEXT:       detect-dead-lanes
 ; GCN-O3-NEXT:       dead-mi-elimination
diff --git a/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll b/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
index 41002382b042f..77072923eb478 100644
--- a/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
+++ b/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
@@ -113,6 +113,7 @@
 ; GCN-O0-NEXT:        Finalize ISel and expand pseudo-instructions
 ; GCN-O0-NEXT:        Local Stack Slot Allocation
 ; GCN-O0-NEXT:        Register Usage Information Propagation
+; GCN-O0-NEXT:        AMDGPU Assign Idx To M0
 ; GCN-O0-NEXT:        Eliminate PHI nodes for register allocation
 ; GCN-O0-NEXT:        SI Lower control flow pseudo instructions
 ; GCN-O0-NEXT:        Two-Address instruction pass
@@ -354,6 +355,7 @@
 ; GCN-O1-NEXT:        Remove dead machine instructions
 ; GCN-O1-NEXT:        SI Shrink Instructions
 ; GCN-O1-NEXT:        Register Usage Information Propagation
+; GCN-O1-NEXT:        AMDGPU Assign Idx To M0
 ; GCN-O1-NEXT:        AMDGPU Prepare AGPR Alloc
 ; GCN-O1-NEXT:        Detect Dead Lanes
 ; GCN-O1-NEXT:        Remove dead machine instructions
@@ -684,6 +686,7 @@
 ; GCN-O1-OPTS-NEXT:        Remove dead machine instructions
 ; GCN-O1-OPTS-NEXT:        SI Shrink Instructions
 ; GCN-O1-OPTS-NEXT:        Register Usage Information Propagation
+; GCN-O1-OPTS-NEXT:        AMDGPU Assign Idx To M0
 ; GCN-O1-OPTS-NEXT:        AMDGPU Prepare AGPR Alloc
 ; GCN-O1-OPTS-NEXT:        Detect Dead Lanes
 ; GCN-O1-OPTS-NEXT:        Remove dead machine instructions
@@ -1018,6 +1021,7 @@
 ; GCN-O2-NEXT:        Remove dead machine instructions
 ; GCN-O2-NEXT:        SI Shrink Instructions
 ; GCN-O2-NEXT:        Register Usage Information Propagation
+; GCN-O2-NEXT:        AMDGPU Assign Idx To M0
 ; GCN-O2-NEXT:        AMDGPU Prepare AGPR Alloc
 ; GCN-O2-NEXT:        Detect Dead Lanes
 ; GCN-O2-NEXT:        Remove dead machine instructions
@@ -1368,6 +1372,7 @@
 ; GCN-O3-NEXT:        Remove dead machine instructions
 ; GCN-O3-NEXT:        SI Shrink Instructions
 ; GCN-O3-NEXT:        Register Usage Information Propagation
+; GCN-O3-NEXT:        AMDGPU Assign Idx To M0
 ; GCN-O3-NEXT:        AMDGPU Prepare AGPR Alloc
 ; GCN-O3-NEXT:        Detect Dead Lanes
 ; GCN-O3-NEXT:        Remove dead machine instructions

>From 12b737f4d394518249791e7ab7a8983557ef1ec4 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Wed, 15 Jul 2026 14:25:19 -0500
Subject: [PATCH 02/35] Waterfall any non-SGPR VGPR-memory index, not just
 virtual

---
 llvm/lib/Target/AMDGPU/SIInstrInfo.cpp | 6 ++++--
 1 file changed, 4 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index a16a6f3fb2d71..e81a31d182d69 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -7788,8 +7788,10 @@ SIInstrInfo::legalizeOperands(MachineInstr &MI,
   // loop that executes the access once per unique index across the wave.
   if (auto *LdStIdx = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
     MachineOperand *Idx = &LdStIdx->getIdxOp();
-    if (Idx->isReg() && Idx->getReg().isVirtual() &&
-        !RI.isSGPRClass(MRI.getRegClass(Idx->getReg())))
+    // Waterfall any non-SGPR index. isSGPRReg handles both virtual and physical
+    // registers, so a physical (non-SGPR) index - not expected here, but still
+    // possible - is made uniform rather than silently skipped.
+    if (Idx->isReg() && !RI.isSGPRReg(MRI, Idx->getReg()))
       CreatedBB = generateWaterFallLoop(*this, MI, {Idx}, MDT);
     return CreatedBB;
   }

>From c3399b87732a311e422c118183fdedc863eb5f72 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Wed, 15 Jul 2026 14:36:32 -0500
Subject: [PATCH 03/35] Do not set a kill flag on the M0 index copy in
 AMDGPUAssignIdxToM0

---
 llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp | 3 ++-
 1 file changed, 2 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
index 8c5de37eb46e5..3bc6709b7c197 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
@@ -55,10 +55,11 @@ static bool assignIdxToM0(MachineFunction &MF) {
       MI.removeOperand(DefIdx);
 
       // Add a copy from the index register to M0 and rewrite MI to read M0.
+      // No kill flag is set on the M0 use: kill flags are deprecated and are a
+      // no-op on the reserved M0 register.
       BuildMI(MBB, &MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY), AMDGPU::M0)
           .add(IdxOp);
       IdxOp.setReg(AMDGPU::M0);
-      IdxOp.setIsKill();
       Changed = true;
     }
   }

>From 00a40607056e17e635c79a7aa2e8291195d7877e Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Thu, 16 Jul 2026 09:07:52 -0500
Subject: [PATCH 04/35] Mask the VGPR-memory movrel base into the addressable
 range

---
 .../Target/AMDGPU/AMDGPULowerVGPREncoding.cpp   | 17 ++++++++++-------
 1 file changed, 10 insertions(+), 7 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
index 4ae470fac3386..bc0754d5e40ac 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
@@ -401,13 +401,14 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
   unsigned Offset = LdSt.getOffsetOp().getImm();
   unsigned NumDwords = LdSt.getBitWidth() / 32;
 
-  // A statically out-of-range dword offset would fold into a base VGPR outside
-  // the addressable register file - an out-of-bounds access of the VGPR
-  // "as memory" (address space 13) region. When the whole file is addressable
-  // the index is allowed to wrap; otherwise it must stay in range.
-#ifndef NDEBUG
+  // A statically out-of-range dword offset is an out-of-bounds (undefined
+  // behavior) access of the VGPR "as memory" (address space 13) region. Rather
+  // than diagnose it or emit an invalid register, mask the base into the
+  // addressable VGPR range below so the access is accepted and verifier-clean,
+  // matching the downstream implementation.
   unsigned NumAddressableVGPRs = ST->getAddressableNumVGPRs(
       MI.getMF()->getInfo<SIMachineFunctionInfo>()->getDynamicVGPRBlockSize());
+#ifndef NDEBUG
   bool AllowOffsetWrap = NumAddressableVGPRs == ST->getTotalNumVGPRs();
   assert((AllowOffsetWrap || Offset + NumDwords <= NumAddressableVGPRs) &&
          "out of bounds VGPR 'as memory' (address space 13) access");
@@ -417,9 +418,11 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
       IsStore ? AMDGPU::V_MOVRELD_B32_e32 : AMDGPU::V_MOVRELS_B32_e32;
 
   // The dword index is (M0 + $offset). Fold $offset into the base register so
-  // each dword i reads/writes VGPR($offset + i) relative to M0.
+  // each dword i reads/writes VGPR($offset + i) relative to M0. Mask the offset
+  // into the addressable range so a statically out-of-bounds offset still
+  // resolves to a valid register.
   for (unsigned i = 0; i < NumDwords; ++i) {
-    Register Base = AMDGPU::VGPR0 + Offset + i;
+    Register Base = AMDGPU::VGPR0 + ((Offset + i) & (NumAddressableVGPRs - 1));
     Register Sub = Data;
     if (NumDwords != 1)
       Sub = TRI->getSubReg(Data, TRI->getSubRegFromChannel(i));

>From 840594950529fe62c2b5f6da5e74c88e073a1f18 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Tue, 28 Jul 2026 00:35:27 +0300
Subject: [PATCH 05/35] Simplify demanded bits

---
 .../Target/AMDGPU/AMDGPULowerVGPREncoding.cpp |  3 +-
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     | 23 +++++++++
 .../as-vgpr-index-demanded-bits.ll            | 51 +++++++++++++++++++
 3 files changed, 75 insertions(+), 2 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
index bc0754d5e40ac..780d1118fecd7 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
@@ -404,8 +404,7 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
   // A statically out-of-range dword offset is an out-of-bounds (undefined
   // behavior) access of the VGPR "as memory" (address space 13) region. Rather
   // than diagnose it or emit an invalid register, mask the base into the
-  // addressable VGPR range below so the access is accepted and verifier-clean,
-  // matching the downstream implementation.
+  // addressable VGPR range below so the access is accepted and verifier-clean.
   unsigned NumAddressableVGPRs = ST->getAddressableNumVGPRs(
       MI.getMF()->getInfo<SIMachineFunctionInfo>()->getDynamicVGPRBlockSize());
 #ifndef NDEBUG
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 2732c91115198..9005135d8ed35 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -19497,6 +19497,29 @@ SDValue SITargetLowering::PerformDAGCombine(SDNode *N,
     return performInsertVectorEltCombine(N, DCI);
   case ISD::FP_ROUND:
     return performFPRoundCombine(N, DCI);
+  case AMDGPUISD::REG_LOAD:
+  case AMDGPUISD::REG_STORE: {
+    const SIMachineFunctionInfo *MFI =
+        DCI.DAG.getMachineFunction().getInfo<SIMachineFunctionInfo>();
+    unsigned NumAddressableVGPRs =
+        Subtarget->getAddressableNumVGPRs(MFI->getDynamicVGPRBlockSize());
+    APInt IndexMask =
+        APInt::getLowBitsSet(32, Log2_32_Ceil(NumAddressableVGPRs));
+
+    unsigned IndexOpIdx = 0;
+    switch (N->getOpcode()) {
+    case AMDGPUISD::REG_LOAD:
+      IndexOpIdx = 1;
+      break;
+    case AMDGPUISD::REG_STORE:
+      IndexOpIdx = 2;
+      break;
+    }
+
+    if (SimplifyDemandedBits(N->getOperand(IndexOpIdx), IndexMask, DCI))
+      return SDValue(N, 0);
+    break;
+  }
   case ISD::LOAD: {
     if (SDValue Widened = widenLoad(cast<LoadSDNode>(N), DCI))
       return Widened;
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll
new file mode 100644
index 0000000000000..1a2689f23a1c4
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll
@@ -0,0 +1,51 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+
+; The VGPR "as memory" (address space 13) dword index only needs enough bits to
+; address all addressable VGPRs, so a redundant high-bit mask feeding the index
+; folds away via the AMDGPUISD::REG_LOAD / REG_STORE SimplifyDemandedBits combine.
+; The incoming index is masked with 0xffff (wider than necessary); on the SDAG
+; path the mask must not survive into the M0 index computation.
+
+define amdgpu_ps i32 @load_masked_index(i32 inreg %arg) {
+; GFX12-SDAG-LABEL: load_masked_index:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-NEXT:    v_readfirstlane_b32 s0, v0
+; GFX12-SDAG-NEXT:    ; return to shader part epilog
+;
+; GFX12-GISEL-LABEL: load_masked_index:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_bfe_u32 m0, s0, 0xe0002
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-GISEL-NEXT:    v_readfirstlane_b32 s0, v0
+; GFX12-GISEL-NEXT:    ; return to shader part epilog
+  %idx = and i32 %arg, 65535
+  %ptr = inttoptr i32 %idx to ptr addrspace(13)
+  %v = load i32, ptr addrspace(13) %ptr
+  ret i32 %v
+}
+
+define amdgpu_ps void @store_masked_index(i32 inreg %arg, i32 %val) {
+; GFX12-SDAG-LABEL: store_masked_index:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-SDAG-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    s_endpgm
+;
+; GFX12-GISEL-LABEL: store_masked_index:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_bfe_u32 m0, s0, 0xe0002
+; GFX12-GISEL-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    s_endpgm
+  %idx = and i32 %arg, 65535
+  %ptr = inttoptr i32 %idx to ptr addrspace(13)
+  store i32 %val, ptr addrspace(13) %ptr
+  ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX12: {{.*}}

>From e026976d0ded5546349967086bca273dbe406f2f Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Wed, 29 Jul 2026 13:56:31 +0300
Subject: [PATCH 06/35] Run AMDGPUAssignIdxToM0 for optnone functions

---
 .../lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp |  5 ++-
 .../AddressSpaceVGPR/as-vgpr-optnone.ll       | 43 +++++++++++++++++++
 2 files changed, 46 insertions(+), 2 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
index 3bc6709b7c197..f2ddf2b5d9bb1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
@@ -76,8 +76,9 @@ class AMDGPUAssignIdxToM0Legacy : public MachineFunctionPass {
   AMDGPUAssignIdxToM0Legacy() : MachineFunctionPass(ID) {}
 
   bool runOnMachineFunction(MachineFunction &MF) override {
-    if (skipFunction(MF.getFunction()))
-      return false;
+    // This is required lowering, not an optimization: without the copy to M0
+    // the movrel that AMDGPULowerVGPREncoding emits later reads a stale index.
+    // It therefore must not be skipped for optnone functions.
     return assignIdxToM0(MF);
   }
 
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll
new file mode 100644
index 0000000000000..cd48756741766
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll
@@ -0,0 +1,43 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12
+; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12
+
+; AMDGPUAssignIdxToM0 is required lowering rather than an optimization: the
+; v_movrel that AMDGPULowerVGPREncoding emits reads the dword index from M0, so
+; without the copy to M0 it reads a stale value. The pass must therefore run
+; even for optnone functions - clang marks every function optnone at -O0 - so
+; the index below has to end up in M0 and not in a plain SGPR.
+
+define i32 @load_i32_optnone(ptr addrspace(13) inreg %p) noinline optnone {
+; GFX12-LABEL: load_i32_optnone:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_mov_b32 s1, 2
+; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT:    s_lshr_b32 m0, s0, s1
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %x = load i32, ptr addrspace(13) %p
+  ret i32 %x
+}
+
+define void @store_i32_optnone(ptr addrspace(13) inreg %p, i32 %v) noinline optnone {
+; GFX12-LABEL: store_i32_optnone:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_mov_b32 s1, 2
+; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT:    s_lshr_b32 m0, s0, s1
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  store i32 %v, ptr addrspace(13) %p
+  ret void
+}

>From ef5b07bf72fd41b70b34df9ec343efe3dac562c6 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Thu, 30 Jul 2026 00:08:17 +0300
Subject: [PATCH 07/35] Stop miscompiling VGPR-memory accesses on subtargets
 without movrel

---
 .../lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp |   4 +
 .../AMDGPU/AMDGPUInstructionSelector.cpp      |  13 +--
 .../Target/AMDGPU/AMDGPULowerVGPREncoding.cpp |  50 +++++++--
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     |  21 ++--
 llvm/lib/Target/AMDGPU/SIInstructions.td      |   7 +-
 .../AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll  | 101 ++++++++++++++++++
 6 files changed, 165 insertions(+), 31 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
index f2ddf2b5d9bb1..b8004ce9925ed 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
@@ -29,6 +29,10 @@ using namespace llvm;
 
 static bool assignIdxToM0(MachineFunction &MF) {
   const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
+
+  // Only movrel takes its index from M0. Subtargets without it index with the
+  // VGPR indexing mode instead, which AMDGPULowerVGPREncoding enables around
+  // the move with s_set_gpr_idx_on, reading the index straight out of its SGPR.
   if (!ST.hasMovrel())
     return false;
 
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
index afc1607ee1b0d..b5455572281c9 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
@@ -4528,14 +4528,11 @@ bool AMDGPUInstructionSelector::selectRegLoadStore(MachineInstr &I) const {
   if (!selectImpl(I, *CoverageInfo))
     return false;
 
-  // On a movrel subtarget the selected V_LOAD_IDX / V_STORE_IDX expands to an
-  // M0-relative move (see AMDGPUAssignIdxToM0 and AMDGPULowerVGPREncoding). Add
-  // an implicit-def of $m0: it records that the eventual move clobbers M0, and
-  // - because an instruction defining a physical register is not hoisted/sunk -
-  // keeps a divergent access pinned inside its waterfall loop.
-  // AMDGPUAssignIdxToM0 removes it when it writes M0 for real.
-  if (!Subtarget->hasMovrel())
-    return true;
+  // The selected V_LOAD_IDX / V_STORE_IDX expands to an M0-relative move (see
+  // AMDGPULowerVGPREncoding), which clobbers M0 whether it indexes with movrel
+  // or with the VGPR indexing mode. Add an implicit-def of $m0 to record that,
+  // and - because an instruction defining a physical register is not
+  // hoisted/sunk - to keep a divergent access pinned inside its waterfall loop.
 
   auto *LdStIdx = cast<AMDGPUMI::VLoadStoreIdxInst>(&*std::prev(II));
   if (LdStIdx->getIdxOp().isReg())
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
index 780d1118fecd7..0974648f1ab83 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
@@ -49,6 +49,7 @@
 #include "SIMachineFunctionInfo.h"
 #include "llvm/ADT/bit.h"
 #include "llvm/CodeGen/MachineBasicBlock.h"
+#include "llvm/CodeGen/MachineInstrBundle.h"
 #include "llvm/Support/Debug.h"
 #include "llvm/Support/MathExtras.h"
 
@@ -181,10 +182,11 @@ class AMDGPULowerVGPREncoding {
   bool runOnMachineInstr(MachineInstr &MI);
 
   /// Lower a VGPR "as memory" (address space 13) indexed load/store pseudo
-  /// (V_LOAD_IDX_B<N> / V_STORE_IDX_B<N>) into a sequence of M0-relative moves
-  /// (v_movrels_b32 for loads, v_movreld_b32 for stores) over the wave's vector
-  /// registers. M0 must already hold the dword index (see AMDGPUAssignIdxToM0).
-  /// This replaces the pseudo, which is erased.
+  /// (V_LOAD_IDX_B<N> / V_STORE_IDX_B<N>) into a sequence of indexed moves over
+  /// the wave's vector registers: v_movrels_b32 / v_movreld_b32 where the
+  /// subtarget has movrel, and v_mov_b32 wrapped in s_set_gpr_idx_on/off where
+  /// it indexes with the VGPR indexing mode. This replaces the pseudo, which is
+  /// erased.
   void lowerLoadStoreIdx(MachineInstr &MI);
 
   /// Compute the mode for a single \p MI given \p Ops operands
@@ -395,9 +397,9 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
   const DebugLoc &DL = MI.getDebugLoc();
   const bool IsStore = LdSt.mayStore();
 
-  // $data is operand 0 of both the load (def) and store (use) pseudos; M0
-  // already holds the dword index (see AMDGPUAssignIdxToM0).
+  // $data is operand 0 of both the load (def) and store (use) pseudos.
   Register Data = LdSt.getDataOp().getReg();
+  MachineOperand &IdxOp = LdSt.getIdxOp();
   unsigned Offset = LdSt.getOffsetOp().getImm();
   unsigned NumDwords = LdSt.getBitWidth() / 32;
 
@@ -413,8 +415,32 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
          "out of bounds VGPR 'as memory' (address space 13) access");
 #endif
 
-  unsigned Opcode =
-      IsStore ? AMDGPU::V_MOVRELD_B32_e32 : AMDGPU::V_MOVRELS_B32_e32;
+  // Subtargets with movrel take the index from M0, which AMDGPUAssignIdxToM0
+  // has already copied it into. The rest have no movrel and index with the VGPR
+  // indexing mode instead: s_set_gpr_idx_on enables it for one operand of the
+  // moves that follow, reading the index straight out of the SGPR holding it,
+  // so no copy is needed there.
+  const bool UseGPRIdxMode = ST->useVGPRIndexMode();
+
+  MachineInstr *SetOn = nullptr;
+  if (UseGPRIdxMode) {
+    SetOn = BuildMI(BB, MI, DL, TII->get(AMDGPU::S_SET_GPR_IDX_ON))
+                .add(IdxOp)
+                .addImm(IsStore ? AMDGPU::VGPRIndexMode::DST_ENABLE
+                                : AMDGPU::VGPRIndexMode::SRC0_ENABLE)
+                .getInstr();
+    SetOn->getOperand(3).setIsUndef();
+  } else {
+    assert(IdxOp.isReg() && IdxOp.getReg() == AMDGPU::M0 &&
+           "movrel index should have been copied into M0");
+  }
+
+  unsigned Opcode;
+  if (UseGPRIdxMode)
+    Opcode = IsStore ? AMDGPU::V_MOV_B32_indirect_write
+                     : AMDGPU::V_MOV_B32_indirect_read;
+  else
+    Opcode = IsStore ? AMDGPU::V_MOVRELD_B32_e32 : AMDGPU::V_MOVRELS_B32_e32;
 
   // The dword index is (M0 + $offset). Fold $offset into the base register so
   // each dword i reads/writes VGPR($offset + i) relative to M0. Mask the offset
@@ -444,6 +470,14 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
       runOnMachineInstr(*Mov);
   }
 
+  // Keep the mode switch and the moves it applies to together, so nothing is
+  // scheduled or spilled in between while indexing is enabled.
+  if (SetOn) {
+    MachineInstr *SetOff =
+        BuildMI(BB, MI, DL, TII->get(AMDGPU::S_SET_GPR_IDX_OFF));
+    finalizeBundle(BB, SetOn->getIterator(), std::next(SetOff->getIterator()));
+  }
+
   MI.eraseFromParent();
 }
 
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 9005135d8ed35..646734e79f260 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -19997,19 +19997,16 @@ void SITargetLowering::AdjustInstrPostInstrSelection(MachineInstr &MI,
     return;
   }
 
-  // A VGPR "as memory" indexed load/store with a register index expands (on
-  // movrel subtargets) to an M0-relative move. Add an implicit-def of $m0: it
-  // records that the eventual move clobbers M0, and - because an instruction
-  // defining a physical register is not hoisted/sunk - keeps a divergent access
-  // pinned inside its waterfall loop. AMDGPUAssignIdxToM0 removes this when it
-  // writes M0 for real.
+  // A VGPR "as memory" indexed load/store with a register index expands to an
+  // M0-relative move, which clobbers M0 whether it indexes with movrel or with
+  // the VGPR indexing mode. Add an implicit-def of $m0: it records that, and -
+  // because an instruction defining a physical register is not hoisted/sunk -
+  // keeps a divergent access pinned inside its waterfall loop.
   if (auto *LdStIdx = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
-    if (getSubtarget()->hasMovrel()) {
-      MachineOperand &IdxOp = LdStIdx->getIdxOp();
-      if (IdxOp.isReg())
-        MI.addOperand(MachineOperand::CreateReg(AMDGPU::M0, /*isDef=*/true,
-                                                /*isImp=*/true));
-    }
+    MachineOperand &IdxOp = LdStIdx->getIdxOp();
+    if (IdxOp.isReg())
+      MI.addOperand(MachineOperand::CreateReg(AMDGPU::M0, /*isDef=*/true,
+                                              /*isImp=*/true));
     return;
   }
 
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index af474d2f297d0..c6a2455ca9cd6 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1069,9 +1069,10 @@ def SI_INDIRECT_DST_V32 : SI_INDIRECT_DST<VReg_1024>;
 // a 32-bit value that may be uniform (SGPR) or divergent (VGPR); $offset is a
 // constant dword offset folded in at selection.
 //
-// AMDGPUAssignIdxToM0 copies $idx into M0 before register allocation, then
-// AMDGPULowerVGPREncoding lowers each into v_movrels_b32 (load) /
-// v_movreld_b32 (store) over the wave's vector registers.
+// AMDGPULowerVGPREncoding lowers each into an M0-relative move over the wave's
+// vector registers: v_movrels_b32 (load) / v_movreld_b32 (store) where the
+// subtarget has movrel, and a v_mov_b32 under the VGPR indexing mode otherwise.
+// It writes $idx to M0 there, beside the move that reads it.
 
 // Populates the VLdStIdxOpcodeInfo searchable table (see AMDGPUMachineInstrs.h
 // and AMDGPU::getVLdStIdxOpcodeInfo*), mapping each pseudo to its bit width and
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
new file mode 100644
index 0000000000000..cebc6ea23c83b
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
@@ -0,0 +1,101 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx942 -o - %s | FileCheck %s --check-prefixes=GFX9,GFX9-SDAG
+; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx942 -o - %s | FileCheck %s --check-prefixes=GFX9,GFX9-GISEL
+; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx90a -filetype=null %s
+; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx90a -filetype=null %s
+
+; The VGPR "as memory" address space (13) on a subtarget that has no movrel.
+; gfx9 indexes with the VGPR indexing mode instead, so AMDGPULowerVGPREncoding
+; wraps the move in s_set_gpr_idx_on / s_set_gpr_idx_off, which takes the dword
+; index straight from the SGPR holding it. Nothing needs to copy the index into
+; M0, so AMDGPUAssignIdxToM0 does nothing on these subtargets.
+;
+; The mode switch and the moves it applies to are bundled, so nothing can be
+; scheduled or spilled between them while indexing is enabled.
+
+define i32 @load_i32(ptr addrspace(13) inreg %p) {
+; GFX9-LABEL: load_i32:
+; GFX9:       ; %bb.0:
+; GFX9-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX9-NEXT:    s_lshr_b32 s0, s0, 2
+; GFX9-NEXT:    s_set_gpr_idx_on s0, gpr_idx(SRC0)
+; GFX9-NEXT:    v_mov_b32_e32 v0, v0
+; GFX9-NEXT:    s_set_gpr_idx_off
+; GFX9-NEXT:    s_setpc_b64 s[30:31]
+  %v = load i32, ptr addrspace(13) %p, align 4
+  ret i32 %v
+}
+
+define void @store_i32(ptr addrspace(13) inreg %p, i32 %v) {
+; GFX9-LABEL: store_i32:
+; GFX9:       ; %bb.0:
+; GFX9-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX9-NEXT:    s_lshr_b32 s0, s0, 2
+; GFX9-NEXT:    s_set_gpr_idx_on s0, gpr_idx(DST)
+; GFX9-NEXT:    v_mov_b32_e32 v0, v0
+; GFX9-NEXT:    s_set_gpr_idx_off
+; GFX9-NEXT:    s_setpc_b64 s[30:31]
+  store i32 %v, ptr addrspace(13) %p, align 4
+  ret void
+}
+
+; More than one dword: every move has to sit inside the same mode switch.
+define i64 @load_i64(ptr addrspace(13) inreg %p) {
+; GFX9-LABEL: load_i64:
+; GFX9:       ; %bb.0:
+; GFX9-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX9-NEXT:    s_lshr_b32 s0, s0, 2
+; GFX9-NEXT:    s_set_gpr_idx_on s0, gpr_idx(SRC0)
+; GFX9-NEXT:    v_mov_b32_e32 v0, v0
+; GFX9-NEXT:    v_mov_b32_e32 v1, v1
+; GFX9-NEXT:    s_set_gpr_idx_off
+; GFX9-NEXT:    s_setpc_b64 s[30:31]
+  %v = load i64, ptr addrspace(13) %p, align 8
+  ret i64 %v
+}
+
+; A divergent pointer still needs the waterfall loop to make the index uniform
+; before the mode switch can use it.
+define i32 @load_i32_divergent(ptr addrspace(13) %p) {
+; GFX9-SDAG-LABEL: load_i32_divergent:
+; GFX9-SDAG:       ; %bb.0:
+; GFX9-SDAG-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX9-SDAG-NEXT:    v_lshrrev_b32_e32 v1, 2, v0
+; GFX9-SDAG-NEXT:    s_mov_b64 s[0:1], exec
+; GFX9-SDAG-NEXT:  .LBB3_1: ; =>This Inner Loop Header: Depth=1
+; GFX9-SDAG-NEXT:    v_readfirstlane_b32 s2, v1
+; GFX9-SDAG-NEXT:    s_nop 1
+; GFX9-SDAG-NEXT:    v_cmp_eq_u32_e32 vcc, s2, v1
+; GFX9-SDAG-NEXT:    s_and_saveexec_b64 vcc, vcc
+; GFX9-SDAG-NEXT:    s_set_gpr_idx_on s2, gpr_idx(SRC0)
+; GFX9-SDAG-NEXT:    v_mov_b32_e32 v0, v0
+; GFX9-SDAG-NEXT:    s_set_gpr_idx_off
+; GFX9-SDAG-NEXT:    ; implicit-def: $vgpr1
+; GFX9-SDAG-NEXT:    s_xor_b64 exec, exec, vcc
+; GFX9-SDAG-NEXT:    s_cbranch_execnz .LBB3_1
+; GFX9-SDAG-NEXT:  ; %bb.2:
+; GFX9-SDAG-NEXT:    s_mov_b64 exec, s[0:1]
+; GFX9-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX9-GISEL-LABEL: load_i32_divergent:
+; GFX9-GISEL:       ; %bb.0:
+; GFX9-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX9-GISEL-NEXT:    v_lshrrev_b32_e32 v1, 2, v0
+; GFX9-GISEL-NEXT:    s_mov_b64 s[0:1], exec
+; GFX9-GISEL-NEXT:  .LBB3_1: ; =>This Inner Loop Header: Depth=1
+; GFX9-GISEL-NEXT:    v_readfirstlane_b32 s4, v1
+; GFX9-GISEL-NEXT:    s_nop 1
+; GFX9-GISEL-NEXT:    v_cmp_eq_u32_e32 vcc, s4, v1
+; GFX9-GISEL-NEXT:    s_and_saveexec_b64 s[2:3], vcc
+; GFX9-GISEL-NEXT:    s_set_gpr_idx_on s4, gpr_idx(SRC0)
+; GFX9-GISEL-NEXT:    v_mov_b32_e32 v0, v0
+; GFX9-GISEL-NEXT:    s_set_gpr_idx_off
+; GFX9-GISEL-NEXT:    ; implicit-def: $vgpr1
+; GFX9-GISEL-NEXT:    s_xor_b64 exec, exec, s[2:3]
+; GFX9-GISEL-NEXT:    s_cbranch_execnz .LBB3_1
+; GFX9-GISEL-NEXT:  ; %bb.2:
+; GFX9-GISEL-NEXT:    s_mov_b64 exec, s[0:1]
+; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %v = load i32, ptr addrspace(13) %p, align 4
+  ret i32 %v
+}

>From f6210c7b0ddfe28504676ecb2c069a59247e3b76 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Thu, 30 Jul 2026 02:15:42 +0300
Subject: [PATCH 08/35] Declare in TableGen that the VGPR-memory pseudos write
 M0

---
 .../lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp | 11 ++------
 .../AMDGPU/AMDGPUInstructionSelector.cpp      | 24 ----------------
 .../Target/AMDGPU/AMDGPUInstructionSelector.h |  1 -
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     | 13 ---------
 llvm/lib/Target/AMDGPU/SIInstructions.td      | 13 +++++----
 .../AddressSpaceVGPR/as-vgpr-divergent.ll     | 15 +++++-----
 .../AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll  | 28 +++++++++----------
 7 files changed, 32 insertions(+), 73 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
index b8004ce9925ed..f3972100574bf 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
@@ -51,14 +51,9 @@ static bool assignIdxToM0(MachineFunction &MF) {
 
       assert(!MI.isBundled());
 
-      // Remove the implicit-def $m0 that instruction selection added (to pin a
-      // divergent access inside its waterfall loop); M0 is written for real
-      // below.
-      int DefIdx = MI.findRegisterDefOperandIdx(AMDGPU::M0, /*TRI=*/nullptr);
-      assert(DefIdx >= 0);
-      MI.removeOperand(DefIdx);
-
-      // Add a copy from the index register to M0 and rewrite MI to read M0.
+      // Add a copy from the index register to M0 and rewrite MI to read M0. The
+      // pseudo goes on declaring that it writes M0: it stands for the whole
+      // sequence, and this copy is the write it describes.
       // No kill flag is set on the M0 use: kill flags are deprecated and are a
       // no-op on the reserved M0 register.
       BuildMI(MBB, &MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY), AMDGPU::M0)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
index b5455572281c9..f06837ff13869 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
@@ -15,7 +15,6 @@
 #include "AMDGPU.h"
 #include "AMDGPUGlobalISelUtils.h"
 #include "AMDGPUInstrInfo.h"
-#include "AMDGPUMachineInstrs.h"
 #include "AMDGPURegisterBankInfo.h"
 #include "SIMachineFunctionInfo.h"
 #include "Utils/AMDGPUBaseInfo.h"
@@ -4521,26 +4520,6 @@ bool AMDGPUInstructionSelector::selectStackRestore(MachineInstr &MI) const {
   return true;
 }
 
-bool AMDGPUInstructionSelector::selectRegLoadStore(MachineInstr &I) const {
-  // Remember where the selected machine instruction will land.
-  MachineBasicBlock::iterator II = std::next(I.getIterator());
-
-  if (!selectImpl(I, *CoverageInfo))
-    return false;
-
-  // The selected V_LOAD_IDX / V_STORE_IDX expands to an M0-relative move (see
-  // AMDGPULowerVGPREncoding), which clobbers M0 whether it indexes with movrel
-  // or with the VGPR indexing mode. Add an implicit-def of $m0 to record that,
-  // and - because an instruction defining a physical register is not
-  // hoisted/sunk - to keep a divergent access pinned inside its waterfall loop.
-
-  auto *LdStIdx = cast<AMDGPUMI::VLoadStoreIdxInst>(&*std::prev(II));
-  if (LdStIdx->getIdxOp().isReg())
-    LdStIdx->addOperand(MachineOperand::CreateReg(AMDGPU::M0, /*isDef=*/true,
-                                                  /*isImp=*/true));
-  return true;
-}
-
 bool AMDGPUInstructionSelector::select(MachineInstr &I) {
 
   if (!I.isPreISelOpcode()) {
@@ -4676,9 +4655,6 @@ bool AMDGPUInstructionSelector::select(MachineInstr &I) {
   case AMDGPU::G_AMDGPU_BVH_INTERSECT_RAY:
   case AMDGPU::G_AMDGPU_BVH8_INTERSECT_RAY:
     return selectBVHIntersectRayIntrinsic(I);
-  case AMDGPU::G_AMDGPU_REG_LOAD:
-  case AMDGPU::G_AMDGPU_REG_STORE:
-    return selectRegLoadStore(I);
   case AMDGPU::G_SBFX:
   case AMDGPU::G_UBFX:
     return selectG_SBFX_UBFX(I);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.h b/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.h
index ab4084472da36..71e815bf122d8 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.h
@@ -91,7 +91,6 @@ class AMDGPUInstructionSelector final : public InstructionSelector {
   bool selectCOPY_VCC_SCC(MachineInstr &I) const;
   bool selectReadAnyLane(MachineInstr &I) const;
   bool selectPHI(MachineInstr &I) const;
-  bool selectRegLoadStore(MachineInstr &I) const;
   bool selectG_TRUNC(MachineInstr &I) const;
   bool selectG_SZA_EXT(MachineInstr &I) const;
   bool selectG_FPEXT(MachineInstr &I) const;
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 646734e79f260..ed63cc7a33d70 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -19997,19 +19997,6 @@ void SITargetLowering::AdjustInstrPostInstrSelection(MachineInstr &MI,
     return;
   }
 
-  // A VGPR "as memory" indexed load/store with a register index expands to an
-  // M0-relative move, which clobbers M0 whether it indexes with movrel or with
-  // the VGPR indexing mode. Add an implicit-def of $m0: it records that, and -
-  // because an instruction defining a physical register is not hoisted/sunk -
-  // keeps a divergent access pinned inside its waterfall loop.
-  if (auto *LdStIdx = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
-    MachineOperand &IdxOp = LdStIdx->getIdxOp();
-    if (IdxOp.isReg())
-      MI.addOperand(MachineOperand::CreateReg(AMDGPU::M0, /*isDef=*/true,
-                                              /*isImp=*/true));
-    return;
-  }
-
   if (TII->isImage(MI))
     TII->enforceOperandRCAlignment(MI, AMDGPU::OpName::vaddr);
 }
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index c6a2455ca9cd6..93246a552aeb5 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1088,10 +1088,11 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
               VReg_512, VReg_1024] in {
   // The units of $idx and $offset are in dwords.
   //
-  // VALU adds an implicit $exec use; together with the implicit-def $m0 added
-  // by AdjustInstrPostInstrSelection (see hasPostISelHook), this keeps a
-  // divergent-index access pinned inside its waterfall loop rather than being
-  // hoisted/sunk out.
+  // The index reaches the hardware through M0, so these define it: either by
+  // writing M0 for v_movrel[sd], or through s_set_gpr_idx_on where the subtarget
+  // indexes with the VGPR indexing mode. Declaring that here also keeps a
+  // divergent-index access pinned inside its waterfall loop, since an
+  // instruction defining a physical register is not hoisted or sunk.
   def V_LOAD_IDX_B#rc.Size : VPseudoInstSI <
     (outs rc:$data),
     (ins SReg_32:$idx, i32imm:$offset)>,
@@ -1100,7 +1101,7 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
       let VALU = 1;
       let UseNamedOperandTable = 1;
       let hasSideEffects = 0;
-      let hasPostISelHook = 1;
+      let Defs = [M0];
   }
   def V_STORE_IDX_B#rc.Size : VPseudoInstSI <
     (outs),
@@ -1110,7 +1111,7 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
       let VALU = 1;
       let UseNamedOperandTable = 1;
       let hasSideEffects = 0;
-      let hasPostISelHook = 1;
+      let Defs = [M0];
   }
 }
 
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
index 28aa0ff7f0fb9..0cde03f7ab74c 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
@@ -47,19 +47,20 @@ define i32 @load_i32(ptr addrspace(13) %p) {
 ; GFX12-GISEL-NEXT:    s_mov_b32 s0, exec_lo
 ; GFX12-GISEL-NEXT:  .LBB0_1: ; =>This Inner Loop Header: Depth=1
 ; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT:    v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT:    s_mov_b32 s1, exec_lo
+; GFX12-GISEL-NEXT:    v_readfirstlane_b32 s1, v0
+; GFX12-GISEL-NEXT:    s_mov_b32 s2, exec_lo
 ; GFX12-GISEL-NEXT:    s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT:    v_cmpx_eq_u32_e32 s2, v0
-; GFX12-GISEL-NEXT:    s_mov_b32 m0, s2
+; GFX12-GISEL-NEXT:    v_cmpx_eq_u32_e32 s1, v0
 ; GFX12-GISEL-NEXT:    ; implicit-def: $vgpr0
-; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v0
 ; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT:    s_xor_b32 exec_lo, exec_lo, s1
+; GFX12-GISEL-NEXT:    s_xor_b32 exec_lo, exec_lo, s2
 ; GFX12-GISEL-NEXT:    s_cbranch_execnz .LBB0_1
 ; GFX12-GISEL-NEXT:  ; %bb.2:
+; GFX12-GISEL-NEXT:    s_mov_b32 m0, s1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
 ; GFX12-GISEL-NEXT:    s_mov_b32 exec_lo, s0
-; GFX12-GISEL-NEXT:    v_add_nc_u32_e32 v0, 1, v1
+; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-GISEL-NEXT:    v_add_nc_u32_e32 v0, 1, v0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
   %x = load i32, ptr addrspace(13) %p
   %y = add i32 %x, 1
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
index cebc6ea23c83b..1d3de91fd8bc2 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
@@ -60,40 +60,40 @@ define i32 @load_i32_divergent(ptr addrspace(13) %p) {
 ; GFX9-SDAG-LABEL: load_i32_divergent:
 ; GFX9-SDAG:       ; %bb.0:
 ; GFX9-SDAG-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX9-SDAG-NEXT:    v_lshrrev_b32_e32 v1, 2, v0
+; GFX9-SDAG-NEXT:    v_lshrrev_b32_e32 v0, 2, v0
 ; GFX9-SDAG-NEXT:    s_mov_b64 s[0:1], exec
 ; GFX9-SDAG-NEXT:  .LBB3_1: ; =>This Inner Loop Header: Depth=1
-; GFX9-SDAG-NEXT:    v_readfirstlane_b32 s2, v1
+; GFX9-SDAG-NEXT:    v_readfirstlane_b32 s2, v0
 ; GFX9-SDAG-NEXT:    s_nop 1
-; GFX9-SDAG-NEXT:    v_cmp_eq_u32_e32 vcc, s2, v1
+; GFX9-SDAG-NEXT:    v_cmp_eq_u32_e32 vcc, s2, v0
 ; GFX9-SDAG-NEXT:    s_and_saveexec_b64 vcc, vcc
-; GFX9-SDAG-NEXT:    s_set_gpr_idx_on s2, gpr_idx(SRC0)
-; GFX9-SDAG-NEXT:    v_mov_b32_e32 v0, v0
-; GFX9-SDAG-NEXT:    s_set_gpr_idx_off
-; GFX9-SDAG-NEXT:    ; implicit-def: $vgpr1
+; GFX9-SDAG-NEXT:    ; implicit-def: $vgpr0
 ; GFX9-SDAG-NEXT:    s_xor_b64 exec, exec, vcc
 ; GFX9-SDAG-NEXT:    s_cbranch_execnz .LBB3_1
 ; GFX9-SDAG-NEXT:  ; %bb.2:
 ; GFX9-SDAG-NEXT:    s_mov_b64 exec, s[0:1]
+; GFX9-SDAG-NEXT:    s_set_gpr_idx_on s2, gpr_idx(SRC0)
+; GFX9-SDAG-NEXT:    v_mov_b32_e32 v0, v0
+; GFX9-SDAG-NEXT:    s_set_gpr_idx_off
 ; GFX9-SDAG-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-GISEL-LABEL: load_i32_divergent:
 ; GFX9-GISEL:       ; %bb.0:
 ; GFX9-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX9-GISEL-NEXT:    v_lshrrev_b32_e32 v1, 2, v0
+; GFX9-GISEL-NEXT:    v_lshrrev_b32_e32 v0, 2, v0
 ; GFX9-GISEL-NEXT:    s_mov_b64 s[0:1], exec
 ; GFX9-GISEL-NEXT:  .LBB3_1: ; =>This Inner Loop Header: Depth=1
-; GFX9-GISEL-NEXT:    v_readfirstlane_b32 s4, v1
+; GFX9-GISEL-NEXT:    v_readfirstlane_b32 s4, v0
 ; GFX9-GISEL-NEXT:    s_nop 1
-; GFX9-GISEL-NEXT:    v_cmp_eq_u32_e32 vcc, s4, v1
+; GFX9-GISEL-NEXT:    v_cmp_eq_u32_e32 vcc, s4, v0
 ; GFX9-GISEL-NEXT:    s_and_saveexec_b64 s[2:3], vcc
-; GFX9-GISEL-NEXT:    s_set_gpr_idx_on s4, gpr_idx(SRC0)
-; GFX9-GISEL-NEXT:    v_mov_b32_e32 v0, v0
-; GFX9-GISEL-NEXT:    s_set_gpr_idx_off
-; GFX9-GISEL-NEXT:    ; implicit-def: $vgpr1
+; GFX9-GISEL-NEXT:    ; implicit-def: $vgpr0
 ; GFX9-GISEL-NEXT:    s_xor_b64 exec, exec, s[2:3]
 ; GFX9-GISEL-NEXT:    s_cbranch_execnz .LBB3_1
 ; GFX9-GISEL-NEXT:  ; %bb.2:
+; GFX9-GISEL-NEXT:    s_set_gpr_idx_on s4, gpr_idx(SRC0)
+; GFX9-GISEL-NEXT:    v_mov_b32_e32 v0, v0
+; GFX9-GISEL-NEXT:    s_set_gpr_idx_off
 ; GFX9-GISEL-NEXT:    s_mov_b64 exec, s[0:1]
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
   %v = load i32, ptr addrspace(13) %p, align 4

>From 8f376da20103706e5a7bcce5fde46f9a614f6380 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Thu, 30 Jul 2026 15:26:49 +0300
Subject: [PATCH 09/35] Explain the undef operands of the VGPR-memory indexed
 moves

---
 .../Target/AMDGPU/AMDGPULowerVGPREncoding.cpp |  14 +-
 .../AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll   | 224 ++++++++++++++++++
 2 files changed, 237 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
index 0974648f1ab83..ec300fefff0fd 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
@@ -446,6 +446,18 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
   // each dword i reads/writes VGPR($offset + i) relative to M0. Mask the offset
   // into the addressable range so a statically out-of-bounds offset still
   // resolves to a valid register.
+  //
+  // A move touches VGPR($offset + i) *plus M0*, which is only known at run
+  // time, so no operand can name the register it really reads or writes.
+  // Operands that name registers as memory rather than a value are therefore
+  // marked undef: the base of every move below, and the stored value as well
+  // when that is itself undef. Liveness of the registers behind this address
+  // space is consequently not expressed here, and correctness relies on nothing
+  // else being allocated to them - which is why frontend use of the address
+  // space is documented as discouraged.
+  const RegState DataFlags = IsStore
+                                 ? getUndefRegState(LdSt.getDataOp().isUndef())
+                                 : RegState::NoFlags;
   for (unsigned i = 0; i < NumDwords; ++i) {
     Register Base = AMDGPU::VGPR0 + ((Offset + i) & (NumAddressableVGPRs - 1));
     Register Sub = Data;
@@ -456,7 +468,7 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
     if (IsStore)
       Mov = BuildMI(BB, MI, DL, TII->get(Opcode))
                 .addReg(Base, RegState::Undef)
-                .addReg(Sub)
+                .addReg(Sub, DataFlags)
                 .getInstr();
     else
       Mov = BuildMI(BB, MI, DL, TII->get(Opcode), Sub)
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll
index 48d90ecaf2eff..34a8d3c2ceb9e 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll
@@ -282,6 +282,230 @@ define void @copy_v8i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in
   ret void
 }
 
+define void @copy_v9i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-SDAG-LABEL: copy_v9i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v9, 0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v8, v8
+; GFX12-SDAG-NEXT:    s_clause 0x2
+; GFX12-SDAG-NEXT:    global_store_b32 v9, v8, s[0:1] offset:32
+; GFX12-SDAG-NEXT:    global_store_b128 v9, v[4:7], s[0:1] offset:16
+; GFX12-SDAG-NEXT:    global_store_b128 v9, v[0:3], s[0:1]
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: copy_v9i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v9, 0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v8, v8
+; GFX12-GISEL-NEXT:    s_clause 0x2
+; GFX12-GISEL-NEXT:    global_store_b128 v9, v[0:3], s[0:1]
+; GFX12-GISEL-NEXT:    global_store_b128 v9, v[4:7], s[0:1] offset:16
+; GFX12-GISEL-NEXT:    global_store_b32 v9, v8, s[0:1] offset:32
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <9 x i32>, ptr addrspace(13) %in
+  store <9 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
+
+define void @copy_v10i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-SDAG-LABEL: copy_v10i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v10, 0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v8, v8
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v9, v9
+; GFX12-SDAG-NEXT:    s_clause 0x2
+; GFX12-SDAG-NEXT:    global_store_b128 v10, v[4:7], s[0:1] offset:16
+; GFX12-SDAG-NEXT:    global_store_b128 v10, v[0:3], s[0:1]
+; GFX12-SDAG-NEXT:    global_store_b64 v10, v[8:9], s[0:1] offset:32
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: copy_v10i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v10, 0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v8, v8
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v9, v9
+; GFX12-GISEL-NEXT:    s_clause 0x2
+; GFX12-GISEL-NEXT:    global_store_b128 v10, v[0:3], s[0:1]
+; GFX12-GISEL-NEXT:    global_store_b128 v10, v[4:7], s[0:1] offset:16
+; GFX12-GISEL-NEXT:    global_store_b64 v10, v[8:9], s[0:1] offset:32
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <10 x i32>, ptr addrspace(13) %in
+  store <10 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
+
+define void @copy_v11i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-SDAG-LABEL: copy_v11i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v11, 0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v8, v8
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v9, v9
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v10, v10
+; GFX12-SDAG-NEXT:    s_clause 0x2
+; GFX12-SDAG-NEXT:    global_store_b128 v11, v[4:7], s[0:1] offset:16
+; GFX12-SDAG-NEXT:    global_store_b128 v11, v[0:3], s[0:1]
+; GFX12-SDAG-NEXT:    global_store_b96 v11, v[8:10], s[0:1] offset:32
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: copy_v11i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v11, 0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v8, v8
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v9, v9
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v10, v10
+; GFX12-GISEL-NEXT:    s_clause 0x2
+; GFX12-GISEL-NEXT:    global_store_b128 v11, v[0:3], s[0:1]
+; GFX12-GISEL-NEXT:    global_store_b128 v11, v[4:7], s[0:1] offset:16
+; GFX12-GISEL-NEXT:    global_store_b96 v11, v[8:10], s[0:1] offset:32
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <11 x i32>, ptr addrspace(13) %in
+  store <11 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
+
+define void @copy_v12i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
+; GFX12-SDAG-LABEL: copy_v12i32:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v12, 0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v8, v8
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v9, v9
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v10, v10
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v11, v11
+; GFX12-SDAG-NEXT:    s_clause 0x2
+; GFX12-SDAG-NEXT:    global_store_b128 v12, v[4:7], s[0:1] offset:16
+; GFX12-SDAG-NEXT:    global_store_b128 v12, v[0:3], s[0:1]
+; GFX12-SDAG-NEXT:    global_store_b128 v12, v[8:11], s[0:1] offset:32
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: copy_v12i32:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s2, 2
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v12, 0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v3, v3
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v4, v4
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v5, v5
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v6, v6
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v7, v7
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v8, v8
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v9, v9
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v10, v10
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v11, v11
+; GFX12-GISEL-NEXT:    s_clause 0x2
+; GFX12-GISEL-NEXT:    global_store_b128 v12, v[0:3], s[0:1]
+; GFX12-GISEL-NEXT:    global_store_b128 v12, v[4:7], s[0:1] offset:16
+; GFX12-GISEL-NEXT:    global_store_b128 v12, v[8:11], s[0:1] offset:32
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %x = load <12 x i32>, ptr addrspace(13) %in
+  store <12 x i32> %x, ptr addrspace(1) %out
+  ret void
+}
+
 define void @copy_v16i32(ptr addrspace(1) inreg %out, ptr addrspace(13) inreg %in) {
 ; GFX12-SDAG-LABEL: copy_v16i32:
 ; GFX12-SDAG:       ; %bb.0:

>From bf6555e7ee8ac1469f9c865728b3b355229edbb7 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Thu, 30 Jul 2026 17:25:11 +0300
Subject: [PATCH 10/35] Give the VGPR-memory moves their own opcodes

---
 .../Target/AMDGPU/AMDGPULowerVGPREncoding.cpp |  3 ++-
 llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp  |  7 ++++++
 llvm/lib/Target/AMDGPU/SIInstrInfo.cpp        | 24 ++++++++-----------
 llvm/lib/Target/AMDGPU/SIInstructions.td      | 15 ++++++++++++
 4 files changed, 34 insertions(+), 15 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
index ec300fefff0fd..bd888c1fee660 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
@@ -440,7 +440,8 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
     Opcode = IsStore ? AMDGPU::V_MOV_B32_indirect_write
                      : AMDGPU::V_MOV_B32_indirect_read;
   else
-    Opcode = IsStore ? AMDGPU::V_MOVRELD_B32_e32 : AMDGPU::V_MOVRELS_B32_e32;
+    Opcode =
+        IsStore ? AMDGPU::V_MOVRELD_B32_as_mem : AMDGPU::V_MOVRELS_B32_as_mem;
 
   // The dword index is (M0 + $offset). Fold $offset into the base register so
   // each dword i reads/writes VGPR($offset + i) relative to M0. Mask the offset
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
index 1758db77d8a6e..d3401c452f2b8 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
@@ -248,6 +248,13 @@ void AMDGPUMCInstLower::lower(const MachineInstr *MI, MCInst &OutMI) const {
              Opcode == AMDGPU::V_FMA_MIX_BF16_t16) {
     lowerT16FmaMixFP16(MI, OutMI);
     return;
+  } else if (Opcode == AMDGPU::V_MOVRELS_B32_as_mem) {
+    // Indexed accesses of the VGPR "as memory" address space use their own
+    // opcodes because they index the register file rather than a tuple; they
+    // encode as the movrel they are named after.
+    Opcode = AMDGPU::V_MOVRELS_B32_e32;
+  } else if (Opcode == AMDGPU::V_MOVRELD_B32_as_mem) {
+    Opcode = AMDGPU::V_MOVRELD_B32_e32;
   }
 
   int MCOpcode = TII->pseudoToMCOpcode(Opcode);
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index e81a31d182d69..47fa902d5f308 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -5909,8 +5909,6 @@ bool SIInstrInfo::verifyInstruction(const MachineInstr &MI,
       Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64) {
     const bool IsDst = Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e32 ||
                        Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64;
-    const bool IsPreRA = !MI.getMF()->getProperties().hasNoVRegs();
-
     const unsigned StaticNumOps =
         Desc.getNumOperands() + Desc.implicit_uses().size();
     const unsigned NumImplicitOps = IsDst ? 2 : 1;
@@ -5919,7 +5917,7 @@ bool SIInstrInfo::verifyInstruction(const MachineInstr &MI,
     // post RA scheduler where the main implicit operand is killed and
     // implicit-defs are added for sub-registers that remain live after this
     // instruction.
-    if (IsPreRA && MI.getNumOperands() < StaticNumOps + NumImplicitOps) {
+    if (MI.getNumOperands() < StaticNumOps + NumImplicitOps) {
       ErrInfo = "missing implicit register operands";
       return false;
     }
@@ -5932,22 +5930,20 @@ bool SIInstrInfo::verifyInstruction(const MachineInstr &MI,
       }
 
       unsigned UseOpIdx;
-      if (IsPreRA && (!MI.isRegTiedToUseOperand(StaticNumOps, &UseOpIdx) ||
-                      UseOpIdx != StaticNumOps + 1)) {
+      if (!MI.isRegTiedToUseOperand(StaticNumOps, &UseOpIdx) ||
+          UseOpIdx != StaticNumOps + 1) {
         ErrInfo = "movrel implicit operands should be tied";
         return false;
       }
     }
 
-    if (IsPreRA) {
-      const MachineOperand &Src0 = MI.getOperand(Src0Idx);
-      const MachineOperand &ImpUse =
-          MI.getOperand(StaticNumOps + NumImplicitOps - 1);
-      if (!ImpUse.isReg() || !ImpUse.isUse() ||
-          !isSubRegOf(RI, ImpUse, IsDst ? *Dst : Src0)) {
-        ErrInfo = "src0 should be subreg of implicit vector use";
-        return false;
-      }
+    const MachineOperand &Src0 = MI.getOperand(Src0Idx);
+    const MachineOperand &ImpUse =
+        MI.getOperand(StaticNumOps + NumImplicitOps - 1);
+    if (!ImpUse.isReg() || !ImpUse.isUse() ||
+        !isSubRegOf(RI, ImpUse, IsDst ? *Dst : Src0)) {
+      ErrInfo = "src0 should be subreg of implicit vector use";
+      return false;
     }
   }
 
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index 93246a552aeb5..b3818e8cbb1eb 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1115,6 +1115,21 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
   }
 }
 
+// Copies of v_movrel[sd]_b32 for the moves the pseudos above expand into.
+//
+// The storage a movrel indexes is normally a register tuple, and the machine
+// verifier requires an implicit use of that tuple naming the register the
+// instruction really touches. Here the storage is the whole register file,
+// which no operand can name, so these carry their own opcodes and leave those
+// rules to the tuple form. AMDGPUMCInstLower maps them back for encoding.
+let VALU = 1, VOP1 = 1, Uses = [M0, EXEC], hasSideEffects = 0,
+    Size = V_MOV_B32_e32.Size in {
+  def V_MOVRELS_B32_as_mem
+      : VPseudoInstSI<(outs VGPR_32:$vdst), (ins VRegSrc_32:$src0)>;
+  def V_MOVRELD_B32_as_mem
+      : VPseudoInstSI<(outs), (ins VGPR_32:$vdst, VSrc_b32:$src0)>;
+}
+
 // Select the REG_LOAD/REG_STORE target nodes into the sized indexed pseudos.
 // The pointer is a byte offset into the file; SIreg_load/SIreg_store carry the
 // dword index (ptr >> 2), and an (add idx, imm) shape folds a constant dword

>From 7ee968116a695c314fa650fe661b20a4181ba807 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Wed, 29 Jul 2026 16:58:56 +0300
Subject: [PATCH 11/35] Pin VGPR-memory indexed accesses to EXEC and mark them
 divergent

---
 .../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp | 10 +++----
 llvm/lib/Target/AMDGPU/SIInstrInfo.cpp        | 13 +++++++++
 .../AddressSpaceVGPR/as-vgpr-divergent.ll     | 15 +++++-----
 .../AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll  | 28 +++++++++----------
 4 files changed, 39 insertions(+), 27 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index 778f6efe6ffce..0cd66044da0b2 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -1697,7 +1697,7 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
     // G_AMDGPU_REG_LOAD/STORE target instructions. Always take the custom path
     // so an unsupported (e.g. sub-dword) access is diagnosed cleanly rather
     // than failing to legalize.
-    Actions.customIf([=](const LegalityQuery &Query) -> bool {
+    Actions.customIf([](const LegalityQuery &Query) -> bool {
       return Query.Types[1].getAddressSpace() == AMDGPUAS::VGPR;
     });
 
@@ -3538,9 +3538,9 @@ static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
     return true;
   }
 
-  const auto PtrAsInt = B.buildPtrToInt(I32, PtrReg);
-  auto Two = B.buildConstant(I32, 2);
-  const auto Index = B.buildLShr(I32, PtrAsInt, Two);
+  const MachineInstrBuilder PtrAsInt = B.buildPtrToInt(I32, PtrReg);
+  MachineInstrBuilder Two = B.buildConstant(I32, 2);
+  const MachineInstrBuilder Index = B.buildLShr(I32, PtrAsInt, Two);
 
   // Normalize the value to i32 / <N x i32> so a selection pattern always
   // exists (e.g. for v4i8).
@@ -3560,7 +3560,7 @@ static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
     B.buildInstr(AMDGPU::G_AMDGPU_REG_LOAD, {ValReg}, {Index.getReg(0)})
         .addMemOperand(&MMO);
   } else {
-    const auto Result =
+    const MachineInstrBuilder Result =
         B.buildInstr(AMDGPU::G_AMDGPU_REG_LOAD, {RegTy}, {Index.getReg(0)})
             .addMemOperand(&MMO);
     B.buildBitcast(ValReg, Result);
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index 47fa902d5f308..fa84743cbdffe 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -308,6 +308,13 @@ bool SIInstrInfo::isSrc1DPPRevOpcode(const GCNSubtarget &ST, uint32_t Opcode) {
 bool SIInstrInfo::resultDependsOnExec(const MachineInstr &MI) const {
   assert(isVALU(MI, /*AllowLDSDMA=*/true));
 
+  // A VGPR "as memory" indexed access reads or writes the per-lane vector
+  // registers of the active lanes, so which lanes are active is part of what it
+  // does. Its implicit use of EXEC must not be treated as ignorable, or the
+  // access could be moved across a write to EXEC.
+  if (isa<AMDGPUMI::VLoadStoreIdxInst>(MI))
+    return true;
+
   // If it is convergent it depends on EXEC.
   if (MI.isConvergent())
     return true;
@@ -11426,6 +11433,12 @@ ValueUniformity SIInstrInfo::getValueUniformity(const MachineInstr &MI) const {
     return ValueUniformity::Default;
   }
 
+  // As above for the generic opcodes, but after instruction selection: an
+  // indexed load reads the wave's per-lane view of its vector registers, so
+  // even a uniform index yields a divergent value.
+  if (isa<AMDGPUMI::VLoadIdxInst>(MI))
+    return ValueUniformity::NeverUniform;
+
   const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
   const AMDGPURegisterBankInfo *RBI = ST.getRegBankInfo();
 
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
index 0cde03f7ab74c..28aa0ff7f0fb9 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
@@ -47,20 +47,19 @@ define i32 @load_i32(ptr addrspace(13) %p) {
 ; GFX12-GISEL-NEXT:    s_mov_b32 s0, exec_lo
 ; GFX12-GISEL-NEXT:  .LBB0_1: ; =>This Inner Loop Header: Depth=1
 ; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT:    v_readfirstlane_b32 s1, v0
-; GFX12-GISEL-NEXT:    s_mov_b32 s2, exec_lo
+; GFX12-GISEL-NEXT:    v_readfirstlane_b32 s2, v0
+; GFX12-GISEL-NEXT:    s_mov_b32 s1, exec_lo
 ; GFX12-GISEL-NEXT:    s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT:    v_cmpx_eq_u32_e32 s1, v0
+; GFX12-GISEL-NEXT:    v_cmpx_eq_u32_e32 s2, v0
+; GFX12-GISEL-NEXT:    s_mov_b32 m0, s2
 ; GFX12-GISEL-NEXT:    ; implicit-def: $vgpr0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v1, v0
 ; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT:    s_xor_b32 exec_lo, exec_lo, s2
+; GFX12-GISEL-NEXT:    s_xor_b32 exec_lo, exec_lo, s1
 ; GFX12-GISEL-NEXT:    s_cbranch_execnz .LBB0_1
 ; GFX12-GISEL-NEXT:  ; %bb.2:
-; GFX12-GISEL-NEXT:    s_mov_b32 m0, s1
-; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
 ; GFX12-GISEL-NEXT:    s_mov_b32 exec_lo, s0
-; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT:    v_add_nc_u32_e32 v0, 1, v0
+; GFX12-GISEL-NEXT:    v_add_nc_u32_e32 v0, 1, v1
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
   %x = load i32, ptr addrspace(13) %p
   %y = add i32 %x, 1
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
index 1d3de91fd8bc2..cebc6ea23c83b 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
@@ -60,40 +60,40 @@ define i32 @load_i32_divergent(ptr addrspace(13) %p) {
 ; GFX9-SDAG-LABEL: load_i32_divergent:
 ; GFX9-SDAG:       ; %bb.0:
 ; GFX9-SDAG-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX9-SDAG-NEXT:    v_lshrrev_b32_e32 v0, 2, v0
+; GFX9-SDAG-NEXT:    v_lshrrev_b32_e32 v1, 2, v0
 ; GFX9-SDAG-NEXT:    s_mov_b64 s[0:1], exec
 ; GFX9-SDAG-NEXT:  .LBB3_1: ; =>This Inner Loop Header: Depth=1
-; GFX9-SDAG-NEXT:    v_readfirstlane_b32 s2, v0
+; GFX9-SDAG-NEXT:    v_readfirstlane_b32 s2, v1
 ; GFX9-SDAG-NEXT:    s_nop 1
-; GFX9-SDAG-NEXT:    v_cmp_eq_u32_e32 vcc, s2, v0
+; GFX9-SDAG-NEXT:    v_cmp_eq_u32_e32 vcc, s2, v1
 ; GFX9-SDAG-NEXT:    s_and_saveexec_b64 vcc, vcc
-; GFX9-SDAG-NEXT:    ; implicit-def: $vgpr0
+; GFX9-SDAG-NEXT:    s_set_gpr_idx_on s2, gpr_idx(SRC0)
+; GFX9-SDAG-NEXT:    v_mov_b32_e32 v0, v0
+; GFX9-SDAG-NEXT:    s_set_gpr_idx_off
+; GFX9-SDAG-NEXT:    ; implicit-def: $vgpr1
 ; GFX9-SDAG-NEXT:    s_xor_b64 exec, exec, vcc
 ; GFX9-SDAG-NEXT:    s_cbranch_execnz .LBB3_1
 ; GFX9-SDAG-NEXT:  ; %bb.2:
 ; GFX9-SDAG-NEXT:    s_mov_b64 exec, s[0:1]
-; GFX9-SDAG-NEXT:    s_set_gpr_idx_on s2, gpr_idx(SRC0)
-; GFX9-SDAG-NEXT:    v_mov_b32_e32 v0, v0
-; GFX9-SDAG-NEXT:    s_set_gpr_idx_off
 ; GFX9-SDAG-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-GISEL-LABEL: load_i32_divergent:
 ; GFX9-GISEL:       ; %bb.0:
 ; GFX9-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX9-GISEL-NEXT:    v_lshrrev_b32_e32 v0, 2, v0
+; GFX9-GISEL-NEXT:    v_lshrrev_b32_e32 v1, 2, v0
 ; GFX9-GISEL-NEXT:    s_mov_b64 s[0:1], exec
 ; GFX9-GISEL-NEXT:  .LBB3_1: ; =>This Inner Loop Header: Depth=1
-; GFX9-GISEL-NEXT:    v_readfirstlane_b32 s4, v0
+; GFX9-GISEL-NEXT:    v_readfirstlane_b32 s4, v1
 ; GFX9-GISEL-NEXT:    s_nop 1
-; GFX9-GISEL-NEXT:    v_cmp_eq_u32_e32 vcc, s4, v0
+; GFX9-GISEL-NEXT:    v_cmp_eq_u32_e32 vcc, s4, v1
 ; GFX9-GISEL-NEXT:    s_and_saveexec_b64 s[2:3], vcc
-; GFX9-GISEL-NEXT:    ; implicit-def: $vgpr0
-; GFX9-GISEL-NEXT:    s_xor_b64 exec, exec, s[2:3]
-; GFX9-GISEL-NEXT:    s_cbranch_execnz .LBB3_1
-; GFX9-GISEL-NEXT:  ; %bb.2:
 ; GFX9-GISEL-NEXT:    s_set_gpr_idx_on s4, gpr_idx(SRC0)
 ; GFX9-GISEL-NEXT:    v_mov_b32_e32 v0, v0
 ; GFX9-GISEL-NEXT:    s_set_gpr_idx_off
+; GFX9-GISEL-NEXT:    ; implicit-def: $vgpr1
+; GFX9-GISEL-NEXT:    s_xor_b64 exec, exec, s[2:3]
+; GFX9-GISEL-NEXT:    s_cbranch_execnz .LBB3_1
+; GFX9-GISEL-NEXT:  ; %bb.2:
 ; GFX9-GISEL-NEXT:    s_mov_b64 exec, s[0:1]
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
   %v = load i32, ptr addrspace(13) %p, align 4

>From 148f572843b6ad84c85e82a0930937bce622a649 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Mon, 3 Aug 2026 19:49:32 +0300
Subject: [PATCH 12/35] Address review: subarch triples, required pass mixin,
 legalizer predicates, redundant VALU

---
 llvm/lib/Target/AMDGPU/AMDGPU.h               |  3 ++-
 .../lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp |  5 +++--
 .../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp | 21 +++++++++----------
 llvm/lib/Target/AMDGPU/SIInstructions.td      |  4 +---
 .../AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll  |  4 ++--
 .../AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll   |  4 ++--
 .../AddressSpaceVGPR/as-vgpr-divergent.ll     |  4 ++--
 .../AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll  |  8 +++----
 .../as-vgpr-index-demanded-bits.ll            |  4 ++--
 .../AddressSpaceVGPR/as-vgpr-inttoptr.ll      |  4 ++--
 .../AddressSpaceVGPR/as-vgpr-optnone.ll       |  4 ++--
 .../AddressSpaceVGPR/as-vgpr-unsupported.ll   |  4 ++--
 12 files changed, 34 insertions(+), 35 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.h b/llvm/lib/Target/AMDGPU/AMDGPU.h
index 87ebce3901845..db909a37697b1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.h
@@ -507,7 +507,8 @@ class AMDGPUMarkLastScratchLoadPass
                         MachineFunctionAnalysisManager &AM);
 };
 
-class AMDGPUAssignIdxToM0Pass : public PassInfoMixin<AMDGPUAssignIdxToM0Pass> {
+class AMDGPUAssignIdxToM0Pass
+    : public RequiredPassInfoMixin<AMDGPUAssignIdxToM0Pass> {
 public:
   PreservedAnalyses run(MachineFunction &MF,
                         MachineFunctionAnalysisManager &MFAM);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
index f3972100574bf..2c651a2d70727 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
@@ -45,9 +45,10 @@ static bool assignIdxToM0(MachineFunction &MF) {
       if (!LdSt)
         continue;
 
+      // The operand class of the index is a register class, so it never holds
+      // an immediate that would have to be moved into M0 separately.
       MachineOperand &IdxOp = LdSt->getIdxOp();
-      if (!IdxOp.isReg())
-        continue;
+      assert(IdxOp.isReg() && "VGPR-memory index must be a register");
 
       assert(!MI.isBundled());
 
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index 0cd66044da0b2..064be0523a221 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -459,6 +459,11 @@ static bool isLoadStoreSizeLegal(const GCNSubtarget &ST,
   if (AS == AMDGPUAS::CONSTANT_ADDRESS_32BIT)
     return false;
 
+  // VGPR ("as memory") accesses are never plain-legal; they are custom-lowered
+  // to G_AMDGPU_REG_LOAD/STORE.
+  if (AS == AMDGPUAS::VGPR)
+    return false;
+
   // Do not handle extending vector loads.
   if (Ty.isVector() && MemSize != RegSize)
     return false;
@@ -551,10 +556,6 @@ static bool loadStoreBitcastWorkaround(const LLT Ty) {
 
 static bool isLoadStoreLegal(const GCNSubtarget &ST, const LegalityQuery &Query) {
   const LLT Ty = Query.Types[0];
-  // VGPR ("as memory") accesses are never plain-legal; they are custom-lowered
-  // to G_AMDGPU_REG_LOAD/STORE.
-  if (Query.Types[1].getAddressSpace() == AMDGPUAS::VGPR)
-    return false;
   return isRegisterType(ST, Ty) && isLoadStoreSizeLegal(ST, Query) &&
          !hasBufferRsrcWorkaround(Ty) && !loadStoreBitcastWorkaround(Ty);
 }
@@ -719,6 +720,7 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
   const LLT RegionPtr = LLT::pointer(AMDGPUAS::REGION_ADDRESS, 32);
   const LLT FlatPtr = LLT::pointer(AMDGPUAS::FLAT_ADDRESS, 64);
   const LLT PrivatePtr = LLT::pointer(AMDGPUAS::PRIVATE_ADDRESS, 32);
+  const LLT VGPRPtr = LLT::pointer(AMDGPUAS::VGPR, 32);
   const LLT BufferFatPtr = LLT::pointer(AMDGPUAS::BUFFER_FAT_POINTER, 160);
   const LLT RsrcPtr = LLT::pointer(AMDGPUAS::BUFFER_RESOURCE, 128);
   const LLT BufferStridedPtr =
@@ -1689,17 +1691,14 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
     // Constant 32-bit is handled by addrspacecasting the 32-bit pointer to
     // 64-bits.
     //
-    // TODO: Should generalize bitcast action into coerce, which will also cover
-    // inserting addrspacecasts.
-    Actions.customIf(typeIs(1, Constant32Ptr));
-
     // VGPR ("as memory") accesses are custom-lowered to the legal
     // G_AMDGPU_REG_LOAD/STORE target instructions. Always take the custom path
     // so an unsupported (e.g. sub-dword) access is diagnosed cleanly rather
     // than failing to legalize.
-    Actions.customIf([](const LegalityQuery &Query) -> bool {
-      return Query.Types[1].getAddressSpace() == AMDGPUAS::VGPR;
-    });
+    //
+    // TODO: Should generalize bitcast action into coerce, which will also cover
+    // inserting addrspacecasts.
+    Actions.customIf(typeInSet(1, {Constant32Ptr, VGPRPtr}));
 
     // Turn any illegal element vectors into something easier to deal
     // with. These will ultimately produce 32-bit scalar shifts to extract the
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index b3818e8cbb1eb..7f27f62fe0a22 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1098,7 +1098,6 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
     (ins SReg_32:$idx, i32imm:$offset)>,
     VLdStIdxOpcodeInfo<rc.Size, 0> {
       let mayLoad = 1;
-      let VALU = 1;
       let UseNamedOperandTable = 1;
       let hasSideEffects = 0;
       let Defs = [M0];
@@ -1108,7 +1107,6 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
     (ins rc:$data, SReg_32:$idx, i32imm:$offset)>,
     VLdStIdxOpcodeInfo<rc.Size, 1> {
       let mayStore = 1;
-      let VALU = 1;
       let UseNamedOperandTable = 1;
       let hasSideEffects = 0;
       let Defs = [M0];
@@ -1122,7 +1120,7 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
 // instruction really touches. Here the storage is the whole register file,
 // which no operand can name, so these carry their own opcodes and leave those
 // rules to the tuple form. AMDGPUMCInstLower maps them back for encoding.
-let VALU = 1, VOP1 = 1, Uses = [M0, EXEC], hasSideEffects = 0,
+let VOP1 = 1, Uses = [M0, EXEC], hasSideEffects = 0,
     Size = V_MOV_B32_e32.Size in {
   def V_MOVRELS_B32_as_mem
       : VPseudoInstSI<(outs VGPR_32:$vdst), (ins VRegSrc_32:$src0)>;
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
index 08d3095de3561..5ff034e4d1893 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
-; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
 
 ; End-to-end lowering of the VGPR "as memory" address space (13) on a
 ; movrel-capable subtarget (gfx12). A load/store of a uniform (SGPR) pointer
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll
index 34a8d3c2ceb9e..d4c916671796c 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
-; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
 
 ; Copy from VGPR "as memory" (address space 13) to global memory across the
 ; range of legal whole-dword access sizes (32 up to 1024 bits). Each uniform
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
index 28aa0ff7f0fb9..4dd453b87bbf2 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
-; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
 
 ; A divergent (per-lane) dword index into VGPR "as memory" (address space 13)
 ; is handled with a waterfall loop: for each unique index across the wave, set
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
index cebc6ea23c83b..ffb673f547dd4 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
@@ -1,8 +1,8 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx942 -o - %s | FileCheck %s --check-prefixes=GFX9,GFX9-SDAG
-; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx942 -o - %s | FileCheck %s --check-prefixes=GFX9,GFX9-GISEL
-; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx90a -filetype=null %s
-; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx90a -filetype=null %s
+; RUN: llc -global-isel=0 -mtriple=amdgpu9.42-- -o - %s | FileCheck %s --check-prefixes=GFX9,GFX9-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu9.42-- -o - %s | FileCheck %s --check-prefixes=GFX9,GFX9-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu9.0a-- -filetype=null %s
+; RUN: llc -global-isel=1 -mtriple=amdgpu9.0a-- -filetype=null %s
 
 ; The VGPR "as memory" address space (13) on a subtarget that has no movrel.
 ; gfx9 indexes with the VGPR indexing mode instead, so AMDGPULowerVGPREncoding
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll
index 1a2689f23a1c4..e1dfb50df7fb9 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
-; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
 
 ; The VGPR "as memory" (address space 13) dword index only needs enough bits to
 ; address all addressable VGPRs, so a redundant high-bit mask feeding the index
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-inttoptr.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-inttoptr.ll
index 393928d359f96..ae759fa010e7a 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-inttoptr.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-inttoptr.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
-; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
 
 ; A constant VGPR "as memory" (address space 13) pointer formed with inttoptr:
 ; the dword index (256 >> 2 = 64) is a compile-time constant, so M0 is set from
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll
index cd48756741766..5d25e4522447b 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12
-; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgcn -mcpu=gfx1200 -o - %s | FileCheck %s --check-prefixes=GFX12
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12
 
 ; AMDGPUAssignIdxToM0 is required lowering rather than an optimization: the
 ; v_movrel that AMDGPULowerVGPREncoding emits reads the dword index from M0, so
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
index 40263e1c975ad..75a744f49f10f 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
@@ -1,5 +1,5 @@
-; RUN: not llc -global-isel=0 -mtriple=amdgcn -mcpu=gfx1200 -filetype=null %s 2>&1 | FileCheck %s
-; RUN: not llc -global-isel=1 -mtriple=amdgcn -mcpu=gfx1200 -filetype=null %s 2>&1 | FileCheck %s
+; RUN: not llc -global-isel=0 -mtriple=amdgpu12.00-- -filetype=null %s 2>&1 | FileCheck %s
+; RUN: not llc -global-isel=1 -mtriple=amdgpu12.00-- -filetype=null %s 2>&1 | FileCheck %s
 
 ; Sub-dword (8/16-bit) accesses of the VGPR "as memory" address space (13) are
 ; not yet implemented. They must be rejected with a clean diagnostic on both

>From 9d81ea779d31ea98ad2d172b41673824ba121edc Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Tue, 4 Aug 2026 01:38:17 +0300
Subject: [PATCH 13/35] Set up M0 for VGPR-memory accesses in finalizeLowering
 instead of a separate pass

---
 llvm/lib/Target/AMDGPU/AMDGPU.h               |  10 --
 .../lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp | 110 ------------------
 .../Target/AMDGPU/AMDGPULowerVGPREncoding.cpp |  11 +-
 llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def |   1 -
 .../lib/Target/AMDGPU/AMDGPUTargetMachine.cpp |   9 --
 llvm/lib/Target/AMDGPU/CMakeLists.txt         |   1 -
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     |  34 ++++++
 llvm/lib/Target/AMDGPU/SIInstructions.td      |   9 +-
 llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll  |   3 -
 llvm/test/CodeGen/AMDGPU/llc-pipeline.ll      |   5 -
 10 files changed, 47 insertions(+), 146 deletions(-)
 delete mode 100644 llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp

diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.h b/llvm/lib/Target/AMDGPU/AMDGPU.h
index db909a37697b1..809540bd05b45 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.h
@@ -288,9 +288,6 @@ extern char &AMDGPURegBankLegalizeLegacyID;
 void initializeAMDGPUMarkLastScratchLoadLegacyPass(PassRegistry &);
 extern char &AMDGPUMarkLastScratchLoadID;
 
-void initializeAMDGPUAssignIdxToM0LegacyPass(PassRegistry &);
-extern char &AMDGPUAssignIdxToM0ID;
-
 void initializeSILowerSGPRSpillsLegacyPass(PassRegistry &);
 extern char &SILowerSGPRSpillsLegacyID;
 
@@ -507,13 +504,6 @@ class AMDGPUMarkLastScratchLoadPass
                         MachineFunctionAnalysisManager &AM);
 };
 
-class AMDGPUAssignIdxToM0Pass
-    : public RequiredPassInfoMixin<AMDGPUAssignIdxToM0Pass> {
-public:
-  PreservedAnalyses run(MachineFunction &MF,
-                        MachineFunctionAnalysisManager &MFAM);
-};
-
 class SIInsertWaitcntsPass
     : public RequiredPassInfoMixin<SIInsertWaitcntsPass> {
 public:
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
deleted file mode 100644
index 2c651a2d70727..0000000000000
--- a/llvm/lib/Target/AMDGPU/AMDGPUAssignIdxToM0.cpp
+++ /dev/null
@@ -1,110 +0,0 @@
-//===- AMDGPUAssignIdxToM0.cpp - Copy VGPR-memory indices to M0 ----------===//
-//
-// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
-// See https://llvm.org/LICENSE.txt for license information.
-// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
-//
-//===----------------------------------------------------------------------===//
-//
-/// \file
-/// Copy the register index of a VGPR "as memory" (address space 13)
-/// V_LOAD_IDX / V_STORE_IDX pseudo into M0, which V_MOVREL[SD] reads when the
-/// pseudo is lowered (see AMDGPULowerVGPREncoding). This runs before register
-/// allocation so the copy to M0 is inserted while the index is still virtual.
-//
-//===----------------------------------------------------------------------===//
-
-#include "AMDGPU.h"
-#include "AMDGPUMachineInstrs.h"
-#include "GCNSubtarget.h"
-#include "SIInstrInfo.h"
-#include "llvm/CodeGen/MachineFunctionPass.h"
-#include "llvm/CodeGen/MachineInstrBuilder.h"
-#include "llvm/CodeGen/MachinePassManager.h"
-#include "llvm/InitializePasses.h"
-
-using namespace llvm;
-
-#define DEBUG_TYPE "amdgpu-assign-idx-to-m0"
-
-static bool assignIdxToM0(MachineFunction &MF) {
-  const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
-
-  // Only movrel takes its index from M0. Subtargets without it index with the
-  // VGPR indexing mode instead, which AMDGPULowerVGPREncoding enables around
-  // the move with s_set_gpr_idx_on, reading the index straight out of its SGPR.
-  if (!ST.hasMovrel())
-    return false;
-
-  const SIInstrInfo *TII = ST.getInstrInfo();
-
-  bool Changed = false;
-  for (MachineBasicBlock &MBB : MF) {
-    for (MachineInstr &MI : MBB) {
-      auto *LdSt = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI);
-      if (!LdSt)
-        continue;
-
-      // The operand class of the index is a register class, so it never holds
-      // an immediate that would have to be moved into M0 separately.
-      MachineOperand &IdxOp = LdSt->getIdxOp();
-      assert(IdxOp.isReg() && "VGPR-memory index must be a register");
-
-      assert(!MI.isBundled());
-
-      // Add a copy from the index register to M0 and rewrite MI to read M0. The
-      // pseudo goes on declaring that it writes M0: it stands for the whole
-      // sequence, and this copy is the write it describes.
-      // No kill flag is set on the M0 use: kill flags are deprecated and are a
-      // no-op on the reserved M0 register.
-      BuildMI(MBB, &MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY), AMDGPU::M0)
-          .add(IdxOp);
-      IdxOp.setReg(AMDGPU::M0);
-      Changed = true;
-    }
-  }
-
-  return Changed;
-}
-
-namespace {
-
-class AMDGPUAssignIdxToM0Legacy : public MachineFunctionPass {
-public:
-  static char ID;
-
-  AMDGPUAssignIdxToM0Legacy() : MachineFunctionPass(ID) {}
-
-  bool runOnMachineFunction(MachineFunction &MF) override {
-    // This is required lowering, not an optimization: without the copy to M0
-    // the movrel that AMDGPULowerVGPREncoding emits later reads a stale index.
-    // It therefore must not be skipped for optnone functions.
-    return assignIdxToM0(MF);
-  }
-
-  void getAnalysisUsage(AnalysisUsage &AU) const override {
-    AU.setPreservesCFG();
-    MachineFunctionPass::getAnalysisUsage(AU);
-  }
-
-  StringRef getPassName() const override { return "AMDGPU Assign Idx To M0"; }
-};
-
-} // end anonymous namespace
-
-PreservedAnalyses
-AMDGPUAssignIdxToM0Pass::run(MachineFunction &MF,
-                             MachineFunctionAnalysisManager &MFAM) {
-  if (!assignIdxToM0(MF))
-    return PreservedAnalyses::all();
-  auto PA = getMachineFunctionPassPreservedAnalyses();
-  PA.preserveSet<CFGAnalyses>();
-  return PA;
-}
-
-char AMDGPUAssignIdxToM0Legacy::ID = 0;
-
-char &llvm::AMDGPUAssignIdxToM0ID = AMDGPUAssignIdxToM0Legacy::ID;
-
-INITIALIZE_PASS(AMDGPUAssignIdxToM0Legacy, DEBUG_TYPE,
-                "AMDGPU Assign Idx To M0", false, false)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
index bd888c1fee660..c065b00e22e55 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
@@ -415,11 +415,12 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
          "out of bounds VGPR 'as memory' (address space 13) access");
 #endif
 
-  // Subtargets with movrel take the index from M0, which AMDGPUAssignIdxToM0
-  // has already copied it into. The rest have no movrel and index with the VGPR
-  // indexing mode instead: s_set_gpr_idx_on enables it for one operand of the
-  // moves that follow, reading the index straight out of the SGPR holding it,
-  // so no copy is needed there.
+  // Subtargets with movrel take the index from M0, which
+  // SITargetLowering::finalizeLowering has already copied it into. The rest
+  // have no movrel and index with the VGPR indexing mode instead:
+  // s_set_gpr_idx_on enables it for one operand of the moves that follow,
+  // reading the index straight out of the SGPR holding it, so no copy is needed
+  // there.
   const bool UseGPRIdxMode = ST->useVGPRIndexMode();
 
   MachineInstr *SetOn = nullptr;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def b/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def
index e7a4435db36c8..372d5f5acab21 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def
@@ -116,7 +116,6 @@ MACHINE_FUNCTION_ANALYSIS("amdgpu-next-use-analysis", AMDGPUNextUseAnalysisPass(
 #define MACHINE_FUNCTION_PASS(NAME, CREATE_PASS)
 #endif
 MACHINE_FUNCTION_PASS("amdgpu-asm-printer", AMDGPUAsmPrinterPass())
-MACHINE_FUNCTION_PASS("amdgpu-assign-idx-to-m0", AMDGPUAssignIdxToM0Pass())
 MACHINE_FUNCTION_PASS("amdgpu-global-isel-divergence-lowering",
                       AMDGPUGlobalISelDivergenceLoweringPass())
 MACHINE_FUNCTION_PASS("amdgpu-insert-delay-alu", AMDGPUInsertDelayAluPass())
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
index a94541b304ee8..72c028edbaef5 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
@@ -702,7 +702,6 @@ extern "C" LLVM_ABI LLVM_EXTERNAL_VISIBILITY void LLVMInitializeAMDGPUTarget() {
   initializeAMDGPURegBankLegalizeLegacyPass(*PR);
   initializeSILowerWWMCopiesLegacyPass(*PR);
   initializeAMDGPUMarkLastScratchLoadLegacyPass(*PR);
-  initializeAMDGPUAssignIdxToM0LegacyPass(*PR);
   initializeSILowerSGPRSpillsLegacyPass(*PR);
   initializeSIFixSGPRCopiesLegacyPass(*PR);
   initializeSIFixVGPRCopiesLegacyPass(*PR);
@@ -1847,10 +1846,6 @@ void GCNPassConfig::addFastRegAlloc() {
 }
 
 void GCNPassConfig::addPreRegAlloc() {
-  // Copy the VGPR "as memory" load/store index into M0 before register
-  // allocation; the movrel emitted later by AMDGPULowerVGPREncoding reads it.
-  addPass(&AMDGPUAssignIdxToM0ID);
-
   if (getOptLevel() != CodeGenOptLevel::None)
     addPass(&AMDGPUPrepareAGPRAllocLegacyID);
   if (getOptLevel() >= CodeGenOptLevel::Default && EnableMachinePipeliner)
@@ -2702,10 +2697,6 @@ Error AMDGPUCodeGenPassBuilder::addOptimizedRegAlloc(PassManagerWrapper &PMW) {
 }
 
 void AMDGPUCodeGenPassBuilder::addPreRegAlloc(PassManagerWrapper &PMW) {
-  // Set up M0 for the movrel that expands a VGPR "as memory" indexed access.
-  // Run before allocation so the index computation coalesces into M0.
-  addMachineFunctionPass(AMDGPUAssignIdxToM0Pass(), PMW);
-
   if (getOptLevel() != CodeGenOptLevel::None)
     addMachineFunctionPass(AMDGPUPrepareAGPRAllocPass(), PMW);
   if (getOptLevel() >= CodeGenOptLevel::Default && EnableMachinePipeliner)
diff --git a/llvm/lib/Target/AMDGPU/CMakeLists.txt b/llvm/lib/Target/AMDGPU/CMakeLists.txt
index cc231e20b13f6..a22cd8b9ab535 100644
--- a/llvm/lib/Target/AMDGPU/CMakeLists.txt
+++ b/llvm/lib/Target/AMDGPU/CMakeLists.txt
@@ -47,7 +47,6 @@ add_llvm_target(AMDGPUCodeGen
   AMDGPUArgumentUsageInfo.cpp
   AMDGPUAsanInstrumentation.cpp
   AMDGPUAsmPrinter.cpp
-  AMDGPUAssignIdxToM0.cpp
   AMDGPUAtomicOptimizer.cpp
   AMDGPUAttributor.cpp
   AMDGPUBarrierLatency.cpp
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index ed63cc7a33d70..3c29c5ec0a370 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -20461,6 +20461,40 @@ void SITargetLowering::finalizeLowering(MachineFunction &MF) const {
 
   Info->limitOccupancy(MF);
 
+  // A VGPR "as memory" indexed access takes its index from M0 on subtargets
+  // with movrel, so copy it there and rewrite the access to read M0, which lets
+  // the index computation coalesce into it. Doing this here rather than while
+  // selecting the access is what makes a divergent index correct: both
+  // selectors make such an index uniform with a waterfall loop - the register
+  // bank legalizer before selection, SIFixSGPRCopies after it - and this runs
+  // after both, so the copy lands inside the loop, next to the per-iteration
+  // index it has to carry. Subtargets without movrel index with the VGPR
+  // indexing mode instead, reading the index straight out of its SGPR.
+  //
+  // TODO: This is only needed because M0 is reserved. Once it is not, the index
+  // can be an ordinary virtual register copy that the coalescer folds away, and
+  // this can go.
+  if (ST.hasMovrel()) {
+    for (MachineBasicBlock &MBB : MF) {
+      for (MachineInstr &MI : MBB) {
+        auto *LdSt = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI);
+        if (!LdSt)
+          continue;
+
+        MachineOperand &IdxOp = LdSt->getIdxOp();
+        assert(IdxOp.isReg() && "VGPR-memory index must be a register");
+        // With GlobalISel this runs twice, once from InstructionSelect and
+        // again from FinalizeISel; the rewrite makes the second run a no-op.
+        if (IdxOp.getReg() == AMDGPU::M0)
+          continue;
+
+        BuildMI(MBB, &MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY), AMDGPU::M0)
+            .addReg(IdxOp.getReg());
+        IdxOp.setReg(AMDGPU::M0);
+      }
+    }
+  }
+
   if (ST.isWave32() && !MF.empty()) {
     for (auto &MBB : MF) {
       for (auto &MI : MBB) {
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index 7f27f62fe0a22..dc25793483fad 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1089,10 +1089,15 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
   // The units of $idx and $offset are in dwords.
   //
   // The index reaches the hardware through M0, so these define it: either by
-  // writing M0 for v_movrel[sd], or through s_set_gpr_idx_on where the subtarget
-  // indexes with the VGPR indexing mode. Declaring that here also keeps a
+  // writing M0 for v_movrel[sd], which SITargetLowering::finalizeLowering
+  // rewrites $idx to, or through s_set_gpr_idx_on where the subtarget indexes
+  // with the VGPR indexing mode. Declaring that here also keeps a
   // divergent-index access pinned inside its waterfall loop, since an
   // instruction defining a physical register is not hoisted or sunk.
+  //
+  // Unlike SI_INDIRECT_SRC/DST and the V_INDIRECT_REG_* pseudos, the storage
+  // indexed here is the whole register file rather than a tuple an operand can
+  // name, so these are memory accesses carrying an MMO instead.
   def V_LOAD_IDX_B#rc.Size : VPseudoInstSI <
     (outs rc:$data),
     (ins SReg_32:$idx, i32imm:$offset)>,
diff --git a/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll b/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll
index 3f3e2b3a314a9..e346eb0a3b072 100644
--- a/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll
+++ b/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll
@@ -68,7 +68,6 @@
 ; GCN-O0-NEXT:   function
 ; GCN-O0-NEXT:     machine-function
 ; GCN-O0-NEXT:       reg-usage-propagation
-; GCN-O0-NEXT:       amdgpu-assign-idx-to-m0
 ; GCN-O0-NEXT:       phi-node-elimination
 ; GCN-O0-NEXT:       si-lower-control-flow
 ; GCN-O0-NEXT:       two-address-instruction
@@ -216,7 +215,6 @@
 ; GCN-O2-NEXT:   function
 ; GCN-O2-NEXT:     machine-function
 ; GCN-O2-NEXT:       reg-usage-propagation
-; GCN-O2-NEXT:       amdgpu-assign-idx-to-m0
 ; GCN-O2-NEXT:       amdgpu-prepare-agpr-alloc
 ; GCN-O2-NEXT:       detect-dead-lanes
 ; GCN-O2-NEXT:       dead-mi-elimination
@@ -405,7 +403,6 @@
 ; GCN-O3-NEXT:   function
 ; GCN-O3-NEXT:     machine-function
 ; GCN-O3-NEXT:       reg-usage-propagation
-; GCN-O3-NEXT:       amdgpu-assign-idx-to-m0
 ; GCN-O3-NEXT:       amdgpu-prepare-agpr-alloc
 ; GCN-O3-NEXT:       detect-dead-lanes
 ; GCN-O3-NEXT:       dead-mi-elimination
diff --git a/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll b/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
index 77072923eb478..41002382b042f 100644
--- a/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
+++ b/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
@@ -113,7 +113,6 @@
 ; GCN-O0-NEXT:        Finalize ISel and expand pseudo-instructions
 ; GCN-O0-NEXT:        Local Stack Slot Allocation
 ; GCN-O0-NEXT:        Register Usage Information Propagation
-; GCN-O0-NEXT:        AMDGPU Assign Idx To M0
 ; GCN-O0-NEXT:        Eliminate PHI nodes for register allocation
 ; GCN-O0-NEXT:        SI Lower control flow pseudo instructions
 ; GCN-O0-NEXT:        Two-Address instruction pass
@@ -355,7 +354,6 @@
 ; GCN-O1-NEXT:        Remove dead machine instructions
 ; GCN-O1-NEXT:        SI Shrink Instructions
 ; GCN-O1-NEXT:        Register Usage Information Propagation
-; GCN-O1-NEXT:        AMDGPU Assign Idx To M0
 ; GCN-O1-NEXT:        AMDGPU Prepare AGPR Alloc
 ; GCN-O1-NEXT:        Detect Dead Lanes
 ; GCN-O1-NEXT:        Remove dead machine instructions
@@ -686,7 +684,6 @@
 ; GCN-O1-OPTS-NEXT:        Remove dead machine instructions
 ; GCN-O1-OPTS-NEXT:        SI Shrink Instructions
 ; GCN-O1-OPTS-NEXT:        Register Usage Information Propagation
-; GCN-O1-OPTS-NEXT:        AMDGPU Assign Idx To M0
 ; GCN-O1-OPTS-NEXT:        AMDGPU Prepare AGPR Alloc
 ; GCN-O1-OPTS-NEXT:        Detect Dead Lanes
 ; GCN-O1-OPTS-NEXT:        Remove dead machine instructions
@@ -1021,7 +1018,6 @@
 ; GCN-O2-NEXT:        Remove dead machine instructions
 ; GCN-O2-NEXT:        SI Shrink Instructions
 ; GCN-O2-NEXT:        Register Usage Information Propagation
-; GCN-O2-NEXT:        AMDGPU Assign Idx To M0
 ; GCN-O2-NEXT:        AMDGPU Prepare AGPR Alloc
 ; GCN-O2-NEXT:        Detect Dead Lanes
 ; GCN-O2-NEXT:        Remove dead machine instructions
@@ -1372,7 +1368,6 @@
 ; GCN-O3-NEXT:        Remove dead machine instructions
 ; GCN-O3-NEXT:        SI Shrink Instructions
 ; GCN-O3-NEXT:        Register Usage Information Propagation
-; GCN-O3-NEXT:        AMDGPU Assign Idx To M0
 ; GCN-O3-NEXT:        AMDGPU Prepare AGPR Alloc
 ; GCN-O3-NEXT:        Detect Dead Lanes
 ; GCN-O3-NEXT:        Remove dead machine instructions

>From 55cbb63f70e230df50ae636ad1b82817a8bbfa38 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Tue, 4 Aug 2026 14:36:35 +0300
Subject: [PATCH 14/35] Round-trip VGPR-memory pointers through flat via a
 synthetic aperture

---
 llvm/docs/AMDGPUUsage.rst                     |   5 +
 .../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp |   7 +-
 llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.cpp  |   2 +
 llvm/lib/Target/AMDGPU/SIDefines.h            |   4 +
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     |  15 +-
 .../AddressSpaceVGPR/as-vgpr-addrspacecast.ll | 164 ++++++++++++++++++
 6 files changed, 191 insertions(+), 6 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll

diff --git a/llvm/docs/AMDGPUUsage.rst b/llvm/docs/AMDGPUUsage.rst
index 9254114e5044f..b41f1af00c5b7 100644
--- a/llvm/docs/AMDGPUUsage.rst
+++ b/llvm/docs/AMDGPUUsage.rst
@@ -1179,6 +1179,11 @@ supported for the ``amdgcn`` target.
   aligned to 2^32 which makes it easier to convert from flat to segment or
   segment to flat.
 
+  *Synthetic apertures* are defined that enable safe roundtrips of pointers
+  from special address spaces through the generic address space. Attempting to
+  dereference generic pointers obtained in this way (using e.g. ``load`` or
+  ``store``) has undefined behavior.
+
   A global address space address has the same value when used as a flat address
   so no conversion is needed.
 
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index 064be0523a221..36a9933b6336e 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -2461,6 +2461,9 @@ bool AMDGPULegalizerInfo::legalizeCustom(
 Register AMDGPULegalizerInfo::getSegmentAperture(unsigned AS,
                                                  MachineRegisterInfo &MRI,
                                                  MachineIRBuilder &B) const {
+  // See SITargetLowering::getSegmentAperture: an address space without an
+  // aperture of its own round-trips through the generic address space using the
+  // shared aperture tagged with its synthetic aperture number.
   unsigned BaseAS = AS;
   unsigned SANum = AMDGPU::getSyntheticApertureNumber(AS);
   if (SANum != AMDGPU::SyntheticAperture::None)
@@ -2616,7 +2619,7 @@ bool AMDGPULegalizerInfo::legalizeAddrSpaceCast(
 
   if (SrcAS == AMDGPUAS::FLAT_ADDRESS &&
       (DestAS == AMDGPUAS::LOCAL_ADDRESS || DestAS == AMDGPUAS::BARRIER ||
-       DestAS == AMDGPUAS::PRIVATE_ADDRESS)) {
+       DestAS == AMDGPUAS::PRIVATE_ADDRESS || DestAS == AMDGPUAS::VGPR)) {
     auto castFlatToLocalOrPrivate = [&](const DstOp &Dst) -> Register {
       if (DestAS == AMDGPUAS::PRIVATE_ADDRESS &&
           ST.hasGloballyAddressableScratch()) {
@@ -2659,7 +2662,7 @@ bool AMDGPULegalizerInfo::legalizeAddrSpaceCast(
 
   if (DestAS == AMDGPUAS::FLAT_ADDRESS &&
       (SrcAS == AMDGPUAS::LOCAL_ADDRESS || SrcAS == AMDGPUAS::BARRIER ||
-       SrcAS == AMDGPUAS::PRIVATE_ADDRESS)) {
+       SrcAS == AMDGPUAS::PRIVATE_ADDRESS || SrcAS == AMDGPUAS::VGPR)) {
     auto castLocalOrPrivateToFlat = [&](const DstOp &Dst) -> Register {
       // Coerce the type of the low half of the result so we can use
       // merge_values.
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.cpp
index 0125a402d0b18..89466128038a1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMemoryUtils.cpp
@@ -35,6 +35,8 @@ unsigned getSyntheticApertureNumber(unsigned AS) {
   switch (AS) {
   case AMDGPUAS::BARRIER:
     return SyntheticAperture::BARRIER;
+  case AMDGPUAS::VGPR:
+    return SyntheticAperture::VGPR;
   default:
     return SyntheticAperture::None;
   }
diff --git a/llvm/lib/Target/AMDGPU/SIDefines.h b/llvm/lib/Target/AMDGPU/SIDefines.h
index 864a84cc8daf4..b481f2e0fa652 100644
--- a/llvm/lib/Target/AMDGPU/SIDefines.h
+++ b/llvm/lib/Target/AMDGPU/SIDefines.h
@@ -1364,10 +1364,14 @@ namespace SyntheticAperture {
 /// bits of the LDS aperture pointer.
 ///
 /// NOTE: This is also documented in AMDGPUUsage.
+///
+/// The addition of new apertures must be coordinated with the architecture
+/// team.
 enum SyntheticAperture {
   None = 0x00000000,
 
   BARRIER = 0x00000001,
+  VGPR = 0x00000003,
 };
 } // namespace SyntheticAperture
 
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 3c29c5ec0a370..7b97a5a0b59c1 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -9445,6 +9445,10 @@ SDValue SITargetLowering::LowerINLINEASM(SDValue Op, SelectionDAG &DAG) const {
 
 SDValue SITargetLowering::getSegmentAperture(unsigned AS, const SDLoc &DL,
                                              SelectionDAG &DAG) const {
+  // An address space that has no aperture of its own round-trips through the
+  // generic address space using a synthetic aperture: the shared aperture with
+  // the aperture number in its low bits. Dereferencing such a generic pointer
+  // is undefined behaviour; the round-trip only has to preserve the value.
   unsigned BaseAS = AS;
   unsigned SANum = AMDGPU::getSyntheticApertureNumber(AS);
   if (SANum != AMDGPU::SyntheticAperture::None)
@@ -9464,6 +9468,7 @@ SDValue SITargetLowering::getBaseSegmentAperture(unsigned AS, const SDLoc &DL,
                                                  SelectionDAG &DAG) const {
   const bool IsLDS = (AS == AMDGPUAS::LOCAL_ADDRESS || AS == AMDGPUAS::BARRIER);
 
+
   if (Subtarget->hasApertureRegs()) {
     const unsigned ApertureRegNo =
         IsLDS ? AMDGPU::SRC_SHARED_BASE : AMDGPU::SRC_PRIVATE_BASE;
@@ -9551,10 +9556,11 @@ SDValue SITargetLowering::lowerADDRSPACECAST(SDValue Op,
 
   SDValue FlatNullPtr = DAG.getConstant(0, SL, MVT::i64);
 
-  // flat -> local/private/barrier
+  // flat -> local/private/barrier/vgpr
   if (SrcAS == AMDGPUAS::FLAT_ADDRESS) {
     if (DestAS == AMDGPUAS::LOCAL_ADDRESS ||
-        DestAS == AMDGPUAS::PRIVATE_ADDRESS || DestAS == AMDGPUAS::BARRIER) {
+        DestAS == AMDGPUAS::PRIVATE_ADDRESS || DestAS == AMDGPUAS::BARRIER ||
+        DestAS == AMDGPUAS::VGPR) {
       SDValue Ptr = DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, Src);
 
       if (DestAS == AMDGPUAS::PRIVATE_ADDRESS &&
@@ -9581,10 +9587,11 @@ SDValue SITargetLowering::lowerADDRSPACECAST(SDValue Op,
     }
   }
 
-  // local/private/barrier -> flat
+  // local/private/barrier/vgpr -> flat
   if (DestAS == AMDGPUAS::FLAT_ADDRESS) {
     if (SrcAS == AMDGPUAS::LOCAL_ADDRESS ||
-        SrcAS == AMDGPUAS::PRIVATE_ADDRESS || SrcAS == AMDGPUAS::BARRIER) {
+        SrcAS == AMDGPUAS::PRIVATE_ADDRESS || SrcAS == AMDGPUAS::BARRIER ||
+        SrcAS == AMDGPUAS::VGPR) {
       SDValue CvtPtr;
       if (SrcAS == AMDGPUAS::PRIVATE_ADDRESS &&
           Subtarget->hasGloballyAddressableScratch()) {
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll
new file mode 100644
index 0000000000000..05efaddedc1e1
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll
@@ -0,0 +1,164 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.50-- -o - %s | FileCheck %s --check-prefixes=GFX1250,GFX1250-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.50-- -o - %s | FileCheck %s --check-prefixes=GFX1250,GFX1250-GISEL
+
+; A pointer to the VGPR "as memory" address space (13) round-trips through the
+; generic address space using a synthetic aperture: the shared aperture with the
+; aperture number in its low bits. The round-trip preserves the value, including
+; the -1 null pointer, but the generic pointer must not be dereferenced.
+
+define ptr @vgpr_to_flat(ptr addrspace(13) %ptr) {
+; GFX12-SDAG-LABEL: vgpr_to_flat:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    v_cmp_ne_u32_e32 vcc_lo, -1, v0
+; GFX12-SDAG-NEXT:    s_mov_b64 s[0:1], src_shared_base
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    s_or_b32 s0, s1, 3
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0) depctr_va_vcc(0)
+; GFX12-SDAG-NEXT:    v_cndmask_b32_e64 v1, 0, s0, vcc_lo
+; GFX12-SDAG-NEXT:    v_cndmask_b32_e32 v0, 0, v0, vcc_lo
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: vgpr_to_flat:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    v_cmp_ne_u32_e32 vcc_lo, -1, v0
+; GFX12-GISEL-NEXT:    s_mov_b64 s[0:1], src_shared_base
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    s_or_b32 s0, s1, 3
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_va_vcc(0)
+; GFX12-GISEL-NEXT:    v_cndmask_b32_e32 v0, 0, v0, vcc_lo
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    v_cndmask_b32_e64 v1, 0, s0, vcc_lo
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-SDAG-LABEL: vgpr_to_flat:
+; GFX1250-SDAG:       ; %bb.0:
+; GFX1250-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-SDAG-NEXT:    v_cmp_ne_u32_e32 vcc_lo, -1, v0
+; GFX1250-SDAG-NEXT:    s_mov_b64 s[0:1], src_shared_base
+; GFX1250-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT:    s_or_b32 s0, s1, 3
+; GFX1250-SDAG-NEXT:    v_cndmask_b32_e64 v1, 0, s0, vcc_lo
+; GFX1250-SDAG-NEXT:    v_cndmask_b32_e32 v0, 0, v0, vcc_lo
+; GFX1250-SDAG-NEXT:    s_set_pc_i64 s[30:31]
+;
+; GFX1250-GISEL-LABEL: vgpr_to_flat:
+; GFX1250-GISEL:       ; %bb.0:
+; GFX1250-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-GISEL-NEXT:    v_cmp_ne_u32_e32 vcc_lo, -1, v0
+; GFX1250-GISEL-NEXT:    s_mov_b64 s[0:1], src_shared_base
+; GFX1250-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT:    s_or_b32 s0, s1, 3
+; GFX1250-GISEL-NEXT:    v_cndmask_b32_e32 v0, 0, v0, vcc_lo
+; GFX1250-GISEL-NEXT:    v_cndmask_b32_e64 v1, 0, s0, vcc_lo
+; GFX1250-GISEL-NEXT:    s_set_pc_i64 s[30:31]
+  %flat = addrspacecast ptr addrspace(13) %ptr to ptr
+  ret ptr %flat
+}
+
+define ptr addrspace(13) @flat_to_vgpr(ptr %flat) {
+; GFX12-LABEL: flat_to_vgpr:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    v_cmp_ne_u64_e32 vcc_lo, 0, v[0:1]
+; GFX12-NEXT:    s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT:    v_cndmask_b32_e32 v0, -1, v0, vcc_lo
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: flat_to_vgpr:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_cmp_ne_u64_e32 vcc_lo, 0, v[0:1]
+; GFX1250-NEXT:    v_cndmask_b32_e32 v0, -1, v0, vcc_lo
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+  %v = addrspacecast ptr %flat to ptr addrspace(13)
+  ret ptr addrspace(13) %v
+}
+
+; With the pointer known to be non-null the null checks are gone.
+
+define ptr @vgpr_to_flat_nonnull(ptr addrspace(13) %ptr) {
+; GFX12-LABEL: vgpr_to_flat_nonnull:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_mov_b64 s[0:1], src_shared_base
+; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT:    s_or_b32 s0, s1, 3
+; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT:    v_mov_b32_e32 v1, s0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: vgpr_to_flat_nonnull:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    s_mov_b64 s[0:1], src_shared_base
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    s_or_b32 s0, s1, 3
+; GFX1250-NEXT:    v_mov_b32_e32 v1, s0
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+  %flat = call ptr @llvm.amdgcn.addrspacecast.nonnull.p0.p13(ptr addrspace(13) %ptr)
+  ret ptr %flat
+}
+
+define ptr addrspace(13) @flat_to_vgpr_nonnull(ptr %flat) {
+; GFX12-LABEL: flat_to_vgpr_nonnull:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: flat_to_vgpr_nonnull:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+  %v = call ptr addrspace(13) @llvm.amdgcn.addrspacecast.nonnull.p13.p0(ptr %flat)
+  ret ptr addrspace(13) %v
+}
+
+define ptr addrspace(13) @vgpr_roundtrip(ptr addrspace(13) %ptr) {
+; GFX12-LABEL: vgpr_roundtrip:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: vgpr_roundtrip:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+  %flat = addrspacecast ptr addrspace(13) %ptr to ptr
+  %v = addrspacecast ptr %flat to ptr addrspace(13)
+  ret ptr addrspace(13) %v
+}

>From 6b9edd3ef84dab85d5f86fdda92828876428f7a2 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Tue, 4 Aug 2026 17:13:17 +0300
Subject: [PATCH 15/35] Declare M0 on the VGPR-memory pseudos only where the
 expansion writes it

---
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     | 69 +++++++++++--------
 llvm/lib/Target/AMDGPU/SIInstructions.td      | 16 ++---
 .../AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll  | 23 +++++++
 3 files changed, 70 insertions(+), 38 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 7b97a5a0b59c1..4a03381d696d4 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -20468,37 +20468,48 @@ void SITargetLowering::finalizeLowering(MachineFunction &MF) const {
 
   Info->limitOccupancy(MF);
 
-  // A VGPR "as memory" indexed access takes its index from M0 on subtargets
-  // with movrel, so copy it there and rewrite the access to read M0, which lets
-  // the index computation coalesce into it. Doing this here rather than while
-  // selecting the access is what makes a divergent index correct: both
-  // selectors make such an index uniform with a waterfall loop - the register
-  // bank legalizer before selection, SIFixSGPRCopies after it - and this runs
-  // after both, so the copy lands inside the loop, next to the per-iteration
-  // index it has to carry. Subtargets without movrel index with the VGPR
-  // indexing mode instead, reading the index straight out of its SGPR.
+  // Give a VGPR "as memory" indexed access its M0 operand, matching how
+  // AMDGPULowerVGPREncoding will expand it. Under the VGPR indexing mode the
+  // expansion emits an s_set_gpr_idx_on, which reads the index out of its SGPR
+  // and clobbers M0, so the access only has to declare that clobber. With
+  // movrel the index has to be in M0, so copy it there and rewrite the access
+  // to read M0, which lets the index computation coalesce into the copy.
   //
-  // TODO: This is only needed because M0 is reserved. Once it is not, the index
-  // can be an ordinary virtual register copy that the coalescer folds away, and
-  // this can go.
-  if (ST.hasMovrel()) {
-    for (MachineBasicBlock &MBB : MF) {
-      for (MachineInstr &MI : MBB) {
-        auto *LdSt = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI);
-        if (!LdSt)
-          continue;
-
-        MachineOperand &IdxOp = LdSt->getIdxOp();
-        assert(IdxOp.isReg() && "VGPR-memory index must be a register");
-        // With GlobalISel this runs twice, once from InstructionSelect and
-        // again from FinalizeISel; the rewrite makes the second run a no-op.
-        if (IdxOp.getReg() == AMDGPU::M0)
-          continue;
-
-        BuildMI(MBB, &MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY), AMDGPU::M0)
-            .addReg(IdxOp.getReg());
-        IdxOp.setReg(AMDGPU::M0);
+  // Doing this here rather than while selecting the access is what makes a
+  // divergent index correct: both selectors make such an index uniform with a
+  // waterfall loop - the register bank legalizer before selection,
+  // SIFixSGPRCopies after it - and this runs after both, so the copy lands
+  // inside the loop, next to the per-iteration index it has to carry.
+  //
+  // TODO: The copy is only needed because M0 is reserved. Once it is not, the
+  // index can be an ordinary virtual register copy that the coalescer folds
+  // away, and this can go.
+  for (MachineBasicBlock &MBB : MF) {
+    for (MachineInstr &MI : MBB) {
+      auto *LdSt = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI);
+      if (!LdSt)
+        continue;
+
+      // With GlobalISel this runs twice, once from InstructionSelect and again
+      // from FinalizeISel, so both forms below have to be idempotent.
+      if (ST.useVGPRIndexMode()) {
+        if (!MI.definesRegister(AMDGPU::M0, TRI))
+          MI.addOperand(MachineOperand::CreateReg(AMDGPU::M0, /*isDef=*/true,
+                                                  /*isImp=*/true));
+        continue;
       }
+
+      MachineOperand &IdxOp = LdSt->getIdxOp();
+      assert(IdxOp.isReg() && "VGPR-memory index must be a register");
+      if (IdxOp.getReg() == AMDGPU::M0)
+        continue;
+
+      BuildMI(MBB, &MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY), AMDGPU::M0)
+          .addReg(IdxOp.getReg());
+      IdxOp.setReg(AMDGPU::M0);
+      // M0 is reserved, and the value copied into it can still be there for a
+      // later access, so this read must not kill it.
+      IdxOp.setIsKill(false);
     }
   }
 
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index dc25793483fad..a5dc5c2dddb07 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1072,7 +1072,6 @@ def SI_INDIRECT_DST_V32 : SI_INDIRECT_DST<VReg_1024>;
 // AMDGPULowerVGPREncoding lowers each into an M0-relative move over the wave's
 // vector registers: v_movrels_b32 (load) / v_movreld_b32 (store) where the
 // subtarget has movrel, and a v_mov_b32 under the VGPR indexing mode otherwise.
-// It writes $idx to M0 there, beside the move that reads it.
 
 // Populates the VLdStIdxOpcodeInfo searchable table (see AMDGPUMachineInstrs.h
 // and AMDGPU::getVLdStIdxOpcodeInfo*), mapping each pseudo to its bit width and
@@ -1088,12 +1087,13 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
               VReg_512, VReg_1024] in {
   // The units of $idx and $offset are in dwords.
   //
-  // The index reaches the hardware through M0, so these define it: either by
-  // writing M0 for v_movrel[sd], which SITargetLowering::finalizeLowering
-  // rewrites $idx to, or through s_set_gpr_idx_on where the subtarget indexes
-  // with the VGPR indexing mode. Declaring that here also keeps a
-  // divergent-index access pinned inside its waterfall loop, since an
-  // instruction defining a physical register is not hoisted or sunk.
+  // The index reaches the hardware through M0, and how it gets there depends on
+  // the subtarget, so SITargetLowering::finalizeLowering fills that in rather
+  // than TableGen: for v_movrel[sd] it copies the index to M0 and rewrites $idx
+  // to it, making the access a plain M0 read; under the VGPR indexing mode the
+  // s_set_gpr_idx_on it expands to clobbers M0 instead, which the access
+  // carries as an implicit def. A divergent-index access is kept inside its
+  // waterfall loop by its implicit use of EXEC (see resultDependsOnExec).
   //
   // Unlike SI_INDIRECT_SRC/DST and the V_INDIRECT_REG_* pseudos, the storage
   // indexed here is the whole register file rather than a tuple an operand can
@@ -1105,7 +1105,6 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
       let mayLoad = 1;
       let UseNamedOperandTable = 1;
       let hasSideEffects = 0;
-      let Defs = [M0];
   }
   def V_STORE_IDX_B#rc.Size : VPseudoInstSI <
     (outs),
@@ -1114,7 +1113,6 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
       let mayStore = 1;
       let UseNamedOperandTable = 1;
       let hasSideEffects = 0;
-      let Defs = [M0];
   }
 }
 
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
index 5ff034e4d1893..295149a115326 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
@@ -297,6 +297,29 @@ define void @copy_i64_aligned(ptr addrspace(13) inreg %dst, ptr addrspace(13) in
   ret void
 }
 
+; An access reads M0 rather than clobbering it, so two accesses at the same
+; index need it written only once.
+
+define i32 @load_i32_twice(ptr addrspace(13) inreg %p) {
+; GFX12-LABEL: load_i32_twice:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT:    v_movrels_b32_e32 v1, v0
+; GFX12-NEXT:    v_add_nc_u32_e32 v0, v0, v1
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %x = load volatile i32, ptr addrspace(13) %p
+  %y = load volatile i32, ptr addrspace(13) %p
+  %z = add i32 %x, %y
+  ret i32 %z
+}
+
 ; Null and poison pointers must be accepted (produce valid code) rather than
 ; crash or fail the machine verifier. The specific null-pointer value is
 ; defined by the parent change that introduces the address space.

>From 6f8819eaac6eb9e6611d39e2fdeb2998a06d536f Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Tue, 25 Aug 2026 13:08:10 -0400
Subject: [PATCH 16/35] Describe the high register bits of the VGPR-memory
 indexed moves

---
 llvm/lib/Target/AMDGPU/SIInstructions.td      |  7 ++-
 .../as-vgpr-high-registers.ll                 | 53 +++++++++++++++++++
 2 files changed, 59 insertions(+), 1 deletion(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-high-registers.ll

diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index a5dc5c2dddb07..d0bc5e09fa743 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1123,8 +1123,13 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
 // instruction really touches. Here the storage is the whole register file,
 // which no operand can name, so these carry their own opcodes and leave those
 // rules to the tuple form. AMDGPUMCInstLower maps them back for encoding.
+// AMDGPULowerVGPREncoding reaches $vdst and $src0 by name to decide which high
+// address bits a subtarget with more than 256 addressable VGPRs needs, so these
+// have to be in the named operand table. Without it that lookup finds nothing,
+// no S_SET_VGPR_MSB is emitted, and a base register at or above 256 is encoded
+// as its low eight bits - a silent access of the wrong register.
 let VOP1 = 1, Uses = [M0, EXEC], hasSideEffects = 0,
-    Size = V_MOV_B32_e32.Size in {
+    UseNamedOperandTable = 1, Size = V_MOV_B32_e32.Size in {
   def V_MOVRELS_B32_as_mem
       : VPseudoInstSI<(outs VGPR_32:$vdst), (ins VRegSrc_32:$src0)>;
   def V_MOVRELD_B32_as_mem
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-high-registers.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-high-registers.ll
new file mode 100644
index 0000000000000..75e726536fe58
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-high-registers.ll
@@ -0,0 +1,53 @@
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.50-- -o - %s | FileCheck %s --check-prefixes=CHECK,SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.50-- -o - %s | FileCheck %s --check-prefixes=CHECK,GISEL
+
+; A subtarget with more than 256 addressable VGPRs encodes a register number's
+; high bits separately, with S_SET_VGPR_MSB. A whole-dword access folds its
+; constant dword offset into the base register of each indexed move, so an
+; access whose base reaches v256 or beyond needs those bits described -
+; otherwise only the low eight bits are encoded and the move silently touches a
+; register 256 lower than the one meant.
+;
+; The moves must therefore be in the named operand table, since that is how
+; AMDGPULowerVGPREncoding finds the operands whose high bits it has to describe.
+;
+; Only SelectionDAG folds the offset into the base; GlobalISel folds it into the
+; index instead and indexes from v0, so it cannot reach a high base this way and
+; needs no mode change. Both are checked, because the difference is the reason
+; this went unnoticed.
+
+; The dword index is %i + 254, so a four-dword access spans v254, v255, v256 and
+; v257 relative to M0.
+; CHECK-LABEL: fold_across_256:
+; SDAG:         v_movrels_b32_e32 v{{[0-9]+}}, v254
+; SDAG-NEXT:    v_movrels_b32_e32 v{{[0-9]+}}, v255
+; SDAG-NEXT:    s_set_vgpr_msb 1
+; SDAG-NEXT:    v_movrels_b32_e32 v{{[0-9]+}}, v0
+; SDAG-NEXT:    v_movrels_b32_e32 v{{[0-9]+}}, v1
+; SDAG:         s_set_vgpr_msb 0x100
+;
+; GISEL:        s_lshl2_add_u32 s0, s0, 0x3f8
+; GISEL:        v_movrels_b32_e32 v{{[0-9]+}}, v0
+; GISEL-NOT:    s_set_vgpr_msb
+define void @fold_across_256(ptr addrspace(1) %out, i32 inreg %i) {
+  %s = shl nuw i32 %i, 2
+  %a = add nuw i32 %s, 1016
+  %p = inttoptr i32 %a to ptr addrspace(13)
+  %v = load <4 x i32>, ptr addrspace(13) %p, align 16
+  store <4 x i32> %v, ptr addrspace(1) %out, align 16
+  ret void
+}
+
+; Entirely below 256, so neither path needs a mode change.
+; CHECK-LABEL: fold_below_256:
+; SDAG:         v_movrels_b32_e32 v{{[0-9]+}}, v4
+; GISEL:        v_movrels_b32_e32 v{{[0-9]+}}, v0
+; CHECK-NOT:    s_set_vgpr_msb
+define void @fold_below_256(ptr addrspace(1) %out, i32 inreg %i) {
+  %s = shl nuw i32 %i, 2
+  %a = add nuw i32 %s, 16
+  %p = inttoptr i32 %a to ptr addrspace(13)
+  %v = load i32, ptr addrspace(13) %p, align 4
+  store i32 %v, ptr addrspace(1) %out, align 4
+  ret void
+}

>From 27fb2254de028508bcf5ae4ae63d28eb22c29e73 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Tue, 25 Aug 2026 13:08:23 -0400
Subject: [PATCH 17/35] Treat a VGPR-memory load as divergent on SelectionDAG
 too

---
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     | 16 +++++--
 .../AddressSpaceVGPR/as-vgpr-divergent.ll     | 45 +++++++++++++++++++
 2 files changed, 58 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 4a03381d696d4..1671341e4866f 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -20893,9 +20893,19 @@ bool SITargetLowering::isSDNodeSourceOfDivergence(const SDNode *N,
   case ISD::LOAD: {
     const LoadSDNode *L = cast<LoadSDNode>(N);
     unsigned AS = L->getAddressSpace();
-    // A flat load may access private memory.
-    return AS == AMDGPUAS::PRIVATE_ADDRESS || AS == AMDGPUAS::FLAT_ADDRESS;
-  }
+    // A flat load may access private memory. A load of the VGPR "as memory"
+    // address space reads this lane's own registers, so it is divergent however
+    // uniform the index is - and it is still an ISD::LOAD until the pre-ISel
+    // combine turns it into a REG_LOAD below.
+    return AS == AMDGPUAS::PRIVATE_ADDRESS || AS == AMDGPUAS::FLAT_ADDRESS ||
+           AS == AMDGPUAS::VGPR;
+  }
+  // The lowered form of the above. Without this the DAG takes the node's
+  // divergence to be that of its operands, so a uniform index makes the loaded
+  // value look uniform, and a consumer that requires a uniform operand gets a
+  // v_readfirstlane - broadcasting one lane's value to the whole wave.
+  case AMDGPUISD::REG_LOAD:
+    return true;
   case ISD::CALLSEQ_END:
     return true;
   case ISD::INTRINSIC_WO_CHAIN:
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
index 4dd453b87bbf2..33ab5a46c37ba 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
@@ -126,5 +126,50 @@ define void @store_i32(ptr addrspace(13) %p, i32 %x) {
   store i32 %y, ptr addrspace(13) %p
   ret void
 }
+; The value read out of the address space is per-lane whatever the index is, so
+; a uniform index does not make it uniform. A consumer that requires a uniform
+; operand must therefore not be handed it directly: doing so inserts a
+; readfirstlane, which broadcasts one lane's value across the whole wave.
+declare i32 @llvm.amdgcn.s.buffer.load.i32(<4 x i32>, i32, i32 immarg)
+
+define i32 @uniform_index_divergent_value(ptr addrspace(13) inreg %p, <4 x i32> inreg %rsrc) {
+; GFX12-SDAG-LABEL: uniform_index_divergent_value:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-SDAG-NEXT:    s_mov_b32 s7, s16
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-SDAG-NEXT:    s_mov_b32 s6, s3
+; GFX12-SDAG-NEXT:    s_mov_b32 s5, s2
+; GFX12-SDAG-NEXT:    s_mov_b32 s4, s1
+; GFX12-SDAG-NEXT:    buffer_load_b32 v0, v0, s[4:7], null offen
+; GFX12-SDAG-NEXT:    s_wait_loadcnt 0x0
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: uniform_index_divergent_value:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-GISEL-NEXT:    s_mov_b32 s4, s1
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    s_mov_b32 s5, s2
+; GFX12-GISEL-NEXT:    s_mov_b32 s6, s3
+; GFX12-GISEL-NEXT:    s_mov_b32 s7, s16
+; GFX12-GISEL-NEXT:    buffer_load_b32 v0, v0, s[4:7], null offen
+; GFX12-GISEL-NEXT:    s_wait_loadcnt 0x0
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %v = load i32, ptr addrspace(13) %p, align 4
+  %r = call i32 @llvm.amdgcn.s.buffer.load.i32(<4 x i32> %rsrc, i32 %v, i32 0)
+  ret i32 %r
+}
+
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; GFX12: {{.*}}

>From 7c585e1afc0bef0088c8dcf6e6567a51faaf42a9 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Tue, 25 Aug 2026 16:02:25 -0400
Subject: [PATCH 18/35] Reject under-aligned whole-dword VGPR-memory accesses

---
 .../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp | 17 +++++---
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     | 19 +++++---
 .../AddressSpaceVGPR/as-vgpr-unsupported.ll   | 43 +++++++++++++++----
 3 files changed, 60 insertions(+), 19 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index 36a9933b6336e..043edec1376ee 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -3523,16 +3523,23 @@ static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
   // integer types rather than plain scalars.
   const LLT I32 = LLT::integer(32);
 
-  // Only whole-dword, non-extending/non-truncating accesses are implemented.
-  // Reject anything else with a diagnostic instead of failing to legalize
-  // (sub-dword support lands in a later change).
+  // Only dword-aligned whole-dword, non-extending/non-truncating accesses are
+  // implemented. Reject anything else with a diagnostic instead of failing to
+  // legalize (sub-dword support lands in a later change).
+  //
+  // The alignment is checked here rather than in the size predicate: the index
+  // built below is the pointer shifted right by two, which discards the low two
+  // bits rather than accounting for them, so an under-aligned access would
+  // silently reach the dword containing the address instead of the bytes asked
+  // for. That is a property of how the address is formed, not of the size.
   if (!isVGPRLoadStoreSizeSupported(MMO.getMemoryType().getSizeInBits(),
-                                    ValSize)) {
+                                    ValSize) ||
+      MMO.getAlign() < Align(4)) {
     const Function &F = B.getMF().getFunction();
     F.getContext().diagnose(DiagnosticInfoUnsupported(
         F,
         "unsupported access of VGPR 'as memory' address space (13); only "
-        "whole-dword loads and stores are implemented",
+        "dword-aligned whole-dword loads and stores are implemented",
         MI.getDebugLoc()));
     if (!IsStore)
       B.buildUndef(ValReg);
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 1671341e4866f..a97910c585a0e 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -13620,17 +13620,17 @@ SDValue SITargetLowering::LowerLoadStoreVGPR(SDValue Op,
   EVT MemVT = MemOp->getMemoryVT();
   unsigned BitWidth = MemVT.getSizeInBits();
 
-  // Only whole-dword, non-extending/non-truncating accesses are implemented.
-  // Reject anything else with a diagnostic (replacing the value with poison)
-  // instead of failing instruction selection. Both callers - operation
-  // legalization and the pre-ISel combine - replace the node with this result,
-  // so the diagnostic is emitted exactly once.
+  // Only dword-aligned whole-dword, non-extending/non-truncating accesses are
+  // implemented. Reject anything else with a diagnostic (replacing the value
+  // with poison) instead of failing instruction selection. Both callers -
+  // operation legalization and the pre-ISel combine - replace the node with
+  // this result, so the diagnostic is emitted exactly once.
   auto reportUnsupported = [&]() -> SDValue {
     const Function &F = DAG.getMachineFunction().getFunction();
     DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
         F,
         "unsupported access of VGPR 'as memory' address space (13); only "
-        "whole-dword loads and stores are implemented",
+        "dword-aligned whole-dword loads and stores are implemented",
         DL.getDebugLoc()));
     if (isa<StoreSDNode>(MemOp))
       return MemOp->getChain();
@@ -13640,6 +13640,13 @@ SDValue SITargetLowering::LowerLoadStoreVGPR(SDValue Op,
 
   if (BitWidth < 32)
     return reportUnsupported();
+
+  // The dword index built below is the pointer shifted right by two, which
+  // discards the low two bits rather than accounting for them, so an
+  // under-aligned access would silently reach the dword containing the address
+  // instead of the bytes asked for.
+  if (MemOp->getAlign() < Align(4))
+    return reportUnsupported();
   if (auto *Load = dyn_cast<LoadSDNode>(MemOp)) {
     if (Load->getExtensionType() != ISD::NON_EXTLOAD)
       return reportUnsupported();
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
index 75a744f49f10f..0cbda8e8ff646 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
@@ -1,31 +1,58 @@
 ; RUN: not llc -global-isel=0 -mtriple=amdgpu12.00-- -filetype=null %s 2>&1 | FileCheck %s
 ; RUN: not llc -global-isel=1 -mtriple=amdgpu12.00-- -filetype=null %s 2>&1 | FileCheck %s
 
-; Sub-dword (8/16-bit) accesses of the VGPR "as memory" address space (13) are
-; not yet implemented. They must be rejected with a clean diagnostic on both
-; SelectionDAG and GlobalISel, rather than failing with "cannot select" /
-; "unable to legalize".
+; Accesses of the VGPR "as memory" address space (13) that are not implemented
+; must be rejected with a clean diagnostic on both SelectionDAG and GlobalISel,
+; rather than failing with "cannot select" / "unable to legalize" - or, worse,
+; silently generating wrong code.
 
-; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only whole-dword loads and stores are implemented
+; Sub-dword (8/16-bit) accesses are not yet implemented; support lands in a
+; later change.
+; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only dword-aligned whole-dword loads and stores are implemented
 define i8 @load_i8(ptr addrspace(13) inreg %p) {
   %x = load i8, ptr addrspace(13) %p
   ret i8 %x
 }
 
-; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only whole-dword loads and stores are implemented
+; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only dword-aligned whole-dword loads and stores are implemented
 define i16 @load_i16(ptr addrspace(13) inreg %p) {
   %x = load i16, ptr addrspace(13) %p
   ret i16 %x
 }
 
-; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only whole-dword loads and stores are implemented
+; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only dword-aligned whole-dword loads and stores are implemented
 define void @store_i8(ptr addrspace(13) inreg %p, i8 %v) {
   store i8 %v, ptr addrspace(13) %p
   ret void
 }
 
-; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only whole-dword loads and stores are implemented
+; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only dword-aligned whole-dword loads and stores are implemented
 define void @store_i16(ptr addrspace(13) inreg %p, i16 %v) {
   store i16 %v, ptr addrspace(13) %p
   ret void
 }
+
+; An access addresses registers by the dword index pointer >> 2, which discards
+; the low two bits rather than accounting for them. An under-aligned one would
+; therefore reach the dword containing the address instead of the bytes asked
+; for - the same code as a correctly aligned access, reading the wrong data with
+; nothing to show for it.
+; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only dword-aligned whole-dword loads and stores are implemented
+define i32 @load_i32_align1(ptr addrspace(13) inreg %p) {
+  %x = load i32, ptr addrspace(13) %p, align 1
+  ret i32 %x
+}
+
+; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only dword-aligned whole-dword loads and stores are implemented
+define void @store_i32_align1(ptr addrspace(13) inreg %p, i32 %v) {
+  store i32 %v, ptr addrspace(13) %p, align 1
+  ret void
+}
+
+; Alignment is required of the pointer, not of the accessed type: a 64-bit
+; access needs only the dword alignment the index computation relies on.
+; CHECK-NOT: in function load_i64_align4
+define i64 @load_i64_align4(ptr addrspace(13) inreg %p) {
+  %x = load i64, ptr addrspace(13) %p, align 4
+  ret i64 %x
+}

>From 522ed240fbdd1d71fb75bb102a3d8ea6fc75bbf0 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Wed, 26 Aug 2026 14:43:11 -0400
Subject: [PATCH 19/35] Cover the divergent index on wave64 and pin the two
 EXEC/uniformity guards

---
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     |   1 -
 .../AddressSpaceVGPR/as-vgpr-divergent.ll     | 125 ++++++++++++++++++
 llvm/unittests/Target/AMDGPU/CMakeLists.txt   |   1 +
 llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp | 125 ++++++++++++++++++
 4 files changed, 251 insertions(+), 1 deletion(-)
 create mode 100644 llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp

diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index a97910c585a0e..1ae383e55dadc 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -9468,7 +9468,6 @@ SDValue SITargetLowering::getBaseSegmentAperture(unsigned AS, const SDLoc &DL,
                                                  SelectionDAG &DAG) const {
   const bool IsLDS = (AS == AMDGPUAS::LOCAL_ADDRESS || AS == AMDGPUAS::BARRIER);
 
-
   if (Subtarget->hasApertureRegs()) {
     const unsigned ApertureRegNo =
         IsLDS ? AMDGPU::SRC_SHARED_BASE : AMDGPU::SRC_PRIVATE_BASE;
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
index 33ab5a46c37ba..e779e6ff4d334 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
@@ -1,11 +1,19 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
 ; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu9.42-- -o - %s | FileCheck %s --check-prefixes=GFX942,GFX942-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu9.42-- -o - %s | FileCheck %s --check-prefixes=GFX942,GFX942-GISEL
 
 ; A divergent (per-lane) dword index into VGPR "as memory" (address space 13)
 ; is handled with a waterfall loop: for each unique index across the wave, set
 ; M0 and do the M0-relative move under a matching-lane EXEC subset. The pointer
 ; arrives in a VGPR (no inreg), so the index (pointer >> 2) is divergent.
+;
+; gfx942 covers the two axes gfx1200 cannot. It is wave64, so the loop runs over
+; a 64-lane mask rather than a 32-lane one, and it indexes with the VGPR indexing
+; mode instead of movrel, so the index reaches the hardware by a different route.
+; gfx1250 cannot stand in for the first: it is wave32 only, and asking it for
+; wave64 makes llc emit no functions at all, which reads as a passing test.
 
 define i32 @load_i32(ptr addrspace(13) %p) {
 ; GFX12-SDAG-LABEL: load_i32:
@@ -61,6 +69,48 @@ define i32 @load_i32(ptr addrspace(13) %p) {
 ; GFX12-GISEL-NEXT:    s_mov_b32 exec_lo, s0
 ; GFX12-GISEL-NEXT:    v_add_nc_u32_e32 v0, 1, v1
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX942-SDAG-LABEL: load_i32:
+; GFX942-SDAG:       ; %bb.0:
+; GFX942-SDAG-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX942-SDAG-NEXT:    v_lshrrev_b32_e32 v0, 2, v0
+; GFX942-SDAG-NEXT:    s_mov_b64 s[0:1], exec
+; GFX942-SDAG-NEXT:  .LBB0_1: ; =>This Inner Loop Header: Depth=1
+; GFX942-SDAG-NEXT:    v_readfirstlane_b32 s2, v0
+; GFX942-SDAG-NEXT:    s_nop 1
+; GFX942-SDAG-NEXT:    v_cmp_eq_u32_e32 vcc, s2, v0
+; GFX942-SDAG-NEXT:    s_and_saveexec_b64 vcc, vcc
+; GFX942-SDAG-NEXT:    s_set_gpr_idx_on s2, gpr_idx(SRC0)
+; GFX942-SDAG-NEXT:    v_mov_b32_e32 v1, v0
+; GFX942-SDAG-NEXT:    s_set_gpr_idx_off
+; GFX942-SDAG-NEXT:    ; implicit-def: $vgpr0
+; GFX942-SDAG-NEXT:    s_xor_b64 exec, exec, vcc
+; GFX942-SDAG-NEXT:    s_cbranch_execnz .LBB0_1
+; GFX942-SDAG-NEXT:  ; %bb.2:
+; GFX942-SDAG-NEXT:    s_mov_b64 exec, s[0:1]
+; GFX942-SDAG-NEXT:    v_add_u32_e32 v0, 1, v1
+; GFX942-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX942-GISEL-LABEL: load_i32:
+; GFX942-GISEL:       ; %bb.0:
+; GFX942-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX942-GISEL-NEXT:    v_lshrrev_b32_e32 v0, 2, v0
+; GFX942-GISEL-NEXT:    s_mov_b64 s[0:1], exec
+; GFX942-GISEL-NEXT:  .LBB0_1: ; =>This Inner Loop Header: Depth=1
+; GFX942-GISEL-NEXT:    v_readfirstlane_b32 s4, v0
+; GFX942-GISEL-NEXT:    s_nop 1
+; GFX942-GISEL-NEXT:    v_cmp_eq_u32_e32 vcc, s4, v0
+; GFX942-GISEL-NEXT:    s_and_saveexec_b64 s[2:3], vcc
+; GFX942-GISEL-NEXT:    s_set_gpr_idx_on s4, gpr_idx(SRC0)
+; GFX942-GISEL-NEXT:    v_mov_b32_e32 v1, v0
+; GFX942-GISEL-NEXT:    s_set_gpr_idx_off
+; GFX942-GISEL-NEXT:    ; implicit-def: $vgpr0
+; GFX942-GISEL-NEXT:    s_xor_b64 exec, exec, s[2:3]
+; GFX942-GISEL-NEXT:    s_cbranch_execnz .LBB0_1
+; GFX942-GISEL-NEXT:  ; %bb.2:
+; GFX942-GISEL-NEXT:    s_mov_b64 exec, s[0:1]
+; GFX942-GISEL-NEXT:    v_add_u32_e32 v0, 1, v1
+; GFX942-GISEL-NEXT:    s_setpc_b64 s[30:31]
   %x = load i32, ptr addrspace(13) %p
   %y = add i32 %x, 1
   ret i32 %y
@@ -122,6 +172,50 @@ define void @store_i32(ptr addrspace(13) %p, i32 %x) {
 ; GFX12-GISEL-NEXT:  ; %bb.2:
 ; GFX12-GISEL-NEXT:    s_mov_b32 exec_lo, s0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX942-SDAG-LABEL: store_i32:
+; GFX942-SDAG:       ; %bb.0:
+; GFX942-SDAG-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX942-SDAG-NEXT:    v_add_u32_e32 v1, 1, v1
+; GFX942-SDAG-NEXT:    v_lshrrev_b32_e32 v0, 2, v0
+; GFX942-SDAG-NEXT:    s_mov_b64 s[0:1], exec
+; GFX942-SDAG-NEXT:  .LBB1_1: ; =>This Inner Loop Header: Depth=1
+; GFX942-SDAG-NEXT:    v_readfirstlane_b32 s2, v0
+; GFX942-SDAG-NEXT:    s_nop 1
+; GFX942-SDAG-NEXT:    v_cmp_eq_u32_e32 vcc, s2, v0
+; GFX942-SDAG-NEXT:    s_and_saveexec_b64 vcc, vcc
+; GFX942-SDAG-NEXT:    s_set_gpr_idx_on s2, gpr_idx(DST)
+; GFX942-SDAG-NEXT:    v_mov_b32_e32 v0, v1
+; GFX942-SDAG-NEXT:    s_set_gpr_idx_off
+; GFX942-SDAG-NEXT:    ; implicit-def: $vgpr0
+; GFX942-SDAG-NEXT:    ; implicit-def: $vgpr1
+; GFX942-SDAG-NEXT:    s_xor_b64 exec, exec, vcc
+; GFX942-SDAG-NEXT:    s_cbranch_execnz .LBB1_1
+; GFX942-SDAG-NEXT:  ; %bb.2:
+; GFX942-SDAG-NEXT:    s_mov_b64 exec, s[0:1]
+; GFX942-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX942-GISEL-LABEL: store_i32:
+; GFX942-GISEL:       ; %bb.0:
+; GFX942-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX942-GISEL-NEXT:    v_add_u32_e32 v1, 1, v1
+; GFX942-GISEL-NEXT:    v_lshrrev_b32_e32 v0, 2, v0
+; GFX942-GISEL-NEXT:    s_mov_b64 s[0:1], exec
+; GFX942-GISEL-NEXT:  .LBB1_1: ; =>This Inner Loop Header: Depth=1
+; GFX942-GISEL-NEXT:    v_readfirstlane_b32 s4, v0
+; GFX942-GISEL-NEXT:    s_nop 1
+; GFX942-GISEL-NEXT:    v_cmp_eq_u32_e32 vcc, s4, v0
+; GFX942-GISEL-NEXT:    s_and_saveexec_b64 s[2:3], vcc
+; GFX942-GISEL-NEXT:    s_set_gpr_idx_on s4, gpr_idx(DST)
+; GFX942-GISEL-NEXT:    v_mov_b32_e32 v0, v1
+; GFX942-GISEL-NEXT:    s_set_gpr_idx_off
+; GFX942-GISEL-NEXT:    ; implicit-def: $vgpr0
+; GFX942-GISEL-NEXT:    ; implicit-def: $vgpr1
+; GFX942-GISEL-NEXT:    s_xor_b64 exec, exec, s[2:3]
+; GFX942-GISEL-NEXT:    s_cbranch_execnz .LBB1_1
+; GFX942-GISEL-NEXT:  ; %bb.2:
+; GFX942-GISEL-NEXT:    s_mov_b64 exec, s[0:1]
+; GFX942-GISEL-NEXT:    s_setpc_b64 s[30:31]
   %y = add i32 %x, 1
   store i32 %y, ptr addrspace(13) %p
   ret void
@@ -166,6 +260,36 @@ define i32 @uniform_index_divergent_value(ptr addrspace(13) inreg %p, <4 x i32>
 ; GFX12-GISEL-NEXT:    buffer_load_b32 v0, v0, s[4:7], null offen
 ; GFX12-GISEL-NEXT:    s_wait_loadcnt 0x0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX942-SDAG-LABEL: uniform_index_divergent_value:
+; GFX942-SDAG:       ; %bb.0:
+; GFX942-SDAG-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX942-SDAG-NEXT:    s_lshr_b32 s0, s0, 2
+; GFX942-SDAG-NEXT:    s_mov_b32 s7, s16
+; GFX942-SDAG-NEXT:    s_mov_b32 s6, s3
+; GFX942-SDAG-NEXT:    s_mov_b32 s5, s2
+; GFX942-SDAG-NEXT:    s_mov_b32 s4, s1
+; GFX942-SDAG-NEXT:    s_set_gpr_idx_on s0, gpr_idx(SRC0)
+; GFX942-SDAG-NEXT:    v_mov_b32_e32 v0, v0
+; GFX942-SDAG-NEXT:    s_set_gpr_idx_off
+; GFX942-SDAG-NEXT:    buffer_load_dword v0, v0, s[4:7], 0 offen
+; GFX942-SDAG-NEXT:    s_waitcnt vmcnt(0)
+; GFX942-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX942-GISEL-LABEL: uniform_index_divergent_value:
+; GFX942-GISEL:       ; %bb.0:
+; GFX942-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX942-GISEL-NEXT:    s_lshr_b32 s0, s0, 2
+; GFX942-GISEL-NEXT:    s_mov_b32 s4, s1
+; GFX942-GISEL-NEXT:    s_mov_b32 s5, s2
+; GFX942-GISEL-NEXT:    s_mov_b32 s6, s3
+; GFX942-GISEL-NEXT:    s_mov_b32 s7, s16
+; GFX942-GISEL-NEXT:    s_set_gpr_idx_on s0, gpr_idx(SRC0)
+; GFX942-GISEL-NEXT:    v_mov_b32_e32 v0, v0
+; GFX942-GISEL-NEXT:    s_set_gpr_idx_off
+; GFX942-GISEL-NEXT:    buffer_load_dword v0, v0, s[4:7], 0 offen
+; GFX942-GISEL-NEXT:    s_waitcnt vmcnt(0)
+; GFX942-GISEL-NEXT:    s_setpc_b64 s[30:31]
   %v = load i32, ptr addrspace(13) %p, align 4
   %r = call i32 @llvm.amdgcn.s.buffer.load.i32(<4 x i32> %rsrc, i32 %v, i32 0)
   ret i32 %r
@@ -173,3 +297,4 @@ define i32 @uniform_index_divergent_value(ptr addrspace(13) inreg %p, <4 x i32>
 
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; GFX12: {{.*}}
+; GFX942: {{.*}}
diff --git a/llvm/unittests/Target/AMDGPU/CMakeLists.txt b/llvm/unittests/Target/AMDGPU/CMakeLists.txt
index 2bb9b6dedba13..27d77a4688fc1 100644
--- a/llvm/unittests/Target/AMDGPU/CMakeLists.txt
+++ b/llvm/unittests/Target/AMDGPU/CMakeLists.txt
@@ -35,4 +35,5 @@ add_llvm_target_unittest(AMDGPUTests
   RCIUpdateReservedRegsTest.cpp
   UniformityAnalysisTest.cpp
   MFMACoExecRules.cpp
+  VGPRAsMemory.cpp
   )
diff --git a/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp b/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp
new file mode 100644
index 0000000000000..01e628a67e9f2
--- /dev/null
+++ b/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp
@@ -0,0 +1,125 @@
+//===--------- llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp --------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// Two properties of the VGPR "as memory" (address space 13) indexed accesses
+// are asserted here rather than in a lit test, because no pass can be made to
+// observe them: these pseudos carry implicit-def $m0, so they define a physical
+// register, and MachineLICM, MachineSink and MachineCSE all decline to touch
+// them for that reason alone. Their safety today is therefore incidental, and
+// the properties below are what it would rest on if that incidental protection
+// ever went away.
+//
+// Each is paired with an ordinary VALU that must answer the other way, so a
+// change that made the query answer uniformly fails here rather than passing
+// vacuously.
+//
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPUUnitTests.h"
+#include "GCNSubtarget.h"
+#include "SIInstrInfo.h"
+#include "llvm/CodeGen/MIRParser/MIRParser.h"
+#include "llvm/CodeGen/MachineModuleInfo.h"
+#include "gtest/gtest.h"
+
+#include "AMDGPUGenSubtargetInfo.inc"
+
+using namespace llvm;
+
+class VGPRAsMemoryTest : public AMDGPUCodeGenTestBase {
+public:
+  void SetUp() override { setUpImpl("amdgpu12.00-amd-", "", ""); }
+};
+
+// An indexed access reads or writes the per-lane vector registers of the active
+// lanes, so which lanes are active is part of what it does. Its implicit use of
+// EXEC must not be reported ignorable: that is what would otherwise let it be
+// hoisted or sunk across a write to EXEC, changing the set of lanes touched.
+TEST_F(VGPRAsMemoryTest, ExecUseIsNotIgnorable) {
+  StringRef MIRString = R"MIR(
+name: exec_use
+body:             |
+  bb.0:
+    liveins: $m0, $vgpr0
+
+    $vgpr1 = V_LOAD_IDX_B32 $m0, 0, implicit $exec :: (load (s32), addrspace 13)
+    V_STORE_IDX_B32 $vgpr0, $m0, 0, implicit $exec :: (store (s32), addrspace 13)
+    $vgpr2 = V_MOV_B32_e32 0, implicit $exec
+    S_ENDPGM 0
+...
+)MIR";
+
+  ASSERT_TRUE(parseMIR(MIRString));
+  MachineFunction &MF = getMF("exec_use");
+  const SIInstrInfo *TII = MF.getSubtarget<GCNSubtarget>().getInstrInfo();
+  MachineBasicBlock *MBB = MF.getBlockNumbered(0);
+
+  auto ExecUseOf = [](const MachineInstr &MI) -> const MachineOperand * {
+    for (const MachineOperand &MO : MI.operands())
+      if (MO.isReg() && MO.isImplicit() && MO.getReg() == AMDGPU::EXEC)
+        return &MO;
+    return nullptr;
+  };
+
+  for (MachineInstr &MI : *MBB) {
+    const MachineOperand *Exec = ExecUseOf(MI);
+    switch (MI.getOpcode()) {
+    case AMDGPU::V_LOAD_IDX_B32:
+    case AMDGPU::V_STORE_IDX_B32:
+      ASSERT_NE(Exec, nullptr) << "indexed access lost its implicit EXEC";
+      EXPECT_FALSE(TII->isIgnorableUse(MI, MI.getOperandNo(Exec)))
+          << "an indexed access may not be moved across a write to EXEC";
+      break;
+    case AMDGPU::V_MOV_B32_e32:
+      // The contrast: a plain lane-wise move produces the same value in every
+      // lane it writes, so its EXEC use really is ignorable.
+      ASSERT_NE(Exec, nullptr);
+      EXPECT_TRUE(TII->isIgnorableUse(MI, MI.getOperandNo(Exec)));
+      break;
+    default:
+      break;
+    }
+  }
+}
+
+// An indexed load reads the wave's per-lane view of its vector registers, so
+// the value is divergent however the index was computed. Reporting it as
+// possibly-uniform would invite a readfirstlane, broadcasting one lane's value
+// across the wave.
+TEST_F(VGPRAsMemoryTest, IndexedLoadIsNeverUniform) {
+  StringRef MIRString = R"MIR(
+name: uniformity
+body:             |
+  bb.0:
+    liveins: $m0
+
+    $vgpr0 = V_LOAD_IDX_B32 $m0, 0, implicit $exec :: (load (s32), addrspace 13)
+    $vgpr1 = V_MOV_B32_e32 0, implicit $exec
+    S_ENDPGM 0
+...
+)MIR";
+
+  ASSERT_TRUE(parseMIR(MIRString));
+  MachineFunction &MF = getMF("uniformity");
+  const SIInstrInfo *TII = MF.getSubtarget<GCNSubtarget>().getInstrInfo();
+  MachineBasicBlock *MBB = MF.getBlockNumbered(0);
+
+  for (MachineInstr &MI : *MBB) {
+    switch (MI.getOpcode()) {
+    case AMDGPU::V_LOAD_IDX_B32:
+      EXPECT_EQ(TII->getValueUniformity(MI), ValueUniformity::NeverUniform);
+      break;
+    case AMDGPU::V_MOV_B32_e32:
+      // The contrast: an ordinary move is only as divergent as its operands.
+      EXPECT_EQ(TII->getValueUniformity(MI), ValueUniformity::Default);
+      break;
+    default:
+      break;
+    }
+  }
+}

>From 3bebcfc2bde853e757f4581c5f9e6cb4be6f9157 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <gt.bercea at gmail.com>
Date: Wed, 9 Sep 2026 16:50:59 -0400
Subject: [PATCH 20/35] Update
 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll

Co-authored-by: Matt Arsenault <arsenm2 at gmail.com>
---
 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
index 295149a115326..c14730105de3b 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
-; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
 
 ; End-to-end lowering of the VGPR "as memory" address space (13) on a
 ; movrel-capable subtarget (gfx12). A load/store of a uniform (SGPR) pointer

>From 6729576b733269286244bac8fc7702c45b9cccfd Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Wed, 9 Sep 2026 19:36:58 -0400
Subject: [PATCH 21/35] Stop reporting two M0-indexed accesses as trivially
 disjoint

---
 llvm/lib/Target/AMDGPU/SIInstrInfo.cpp        | 10 +++++
 llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp | 43 +++++++++++++++++++
 2 files changed, 53 insertions(+)

diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index fa84743cbdffe..cb72c11403088 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -675,6 +675,16 @@ bool SIInstrInfo::getMemOperandsWithOffsetWidth(
     BaseOp = &LdStIdx->getIdxOp();
     OffsetOp = &LdStIdx->getOffsetOp();
 
+    // Callers compare two accesses by base operand and constant offset, and
+    // treat identical bases as the same address. That only holds while the base
+    // names a value. On a movrel subtarget every one of these reads M0, so two
+    // accesses with unrelated indices compare as the same base and would be
+    // declared disjoint on their offsets alone; M0 can also be redefined
+    // between them, which the comparison does not look for. Report such an
+    // access as opaque instead.
+    if (!BaseOp->isReg() || !BaseOp->getReg().isVirtual())
+      return false;
+
     BaseOps.push_back(BaseOp);
     Offset = OffsetOp->getImm() * 4; // Offset has units of dwords.
 
diff --git a/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp b/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp
index 01e628a67e9f2..a55b78a1de628 100644
--- a/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp
+++ b/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp
@@ -123,3 +123,46 @@ body:             |
     }
   }
 }
+
+// Two whole-dword accesses whose index operand is M0 - the form every access
+// has on a movrel subtarget - with M0 redefined between them. The dwords they
+// touch are (first M0)+1 and (second M0)+0, which are the same dword whenever
+// the second index is one more than the first, so they may alias. Disjointness
+// is decided from the base operand and the constant offset, and both bases are
+// literally $m0, so nothing in that comparison can tell the two M0 values
+// apart.
+TEST_F(VGPRAsMemoryTest, M0IndexedAccessesAcrossAM0RedefMayAlias) {
+  StringRef MIRString = R"MIR(
+name: m0_redef
+body:             |
+  bb.0:
+    liveins: $sgpr0, $sgpr1, $vgpr0
+
+    $m0 = COPY $sgpr0
+    $vgpr1 = V_LOAD_IDX_B32 $m0, 1, implicit $exec :: (load (s32), addrspace 13)
+    $m0 = COPY $sgpr1
+    V_STORE_IDX_B32 $vgpr0, $m0, 0, implicit $exec :: (store (s32), addrspace 13)
+    S_ENDPGM 0
+...
+)MIR";
+
+  ASSERT_TRUE(parseMIR(MIRString));
+  MachineFunction &MF = getMF("m0_redef");
+  const SIInstrInfo *TII = MF.getSubtarget<GCNSubtarget>().getInstrInfo();
+  MachineBasicBlock *MBB = MF.getBlockNumbered(0);
+
+  const MachineInstr *Load = nullptr;
+  const MachineInstr *Store = nullptr;
+  for (MachineInstr &MI : *MBB) {
+    if (MI.getOpcode() == AMDGPU::V_LOAD_IDX_B32)
+      Load = &MI;
+    else if (MI.getOpcode() == AMDGPU::V_STORE_IDX_B32)
+      Store = &MI;
+  }
+  ASSERT_NE(Load, nullptr);
+  ASSERT_NE(Store, nullptr);
+
+  EXPECT_FALSE(TII->areMemAccessesTriviallyDisjoint(*Load, *Store))
+      << "M0 is redefined between these accesses, so their dword indices are "
+         "unrelated and they must not be reported disjoint";
+}

>From fdd45e9a3ec6718b324b5b9de798d16a8465e144 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Wed, 9 Sep 2026 19:39:04 -0400
Subject: [PATCH 22/35] Set up M0 for VGPR-memory accesses in a custom inserter

---
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp | 86 +++++++++++------------
 llvm/lib/Target/AMDGPU/SIInstructions.td  | 19 +++--
 2 files changed, 54 insertions(+), 51 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 1ae383e55dadc..b41acf72541ee 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -6958,6 +6958,47 @@ SITargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI,
   MachineRegisterInfo &MRI = MF->getRegInfo();
   const DebugLoc &DL = MI.getDebugLoc();
 
+  // Give a VGPR "as memory" indexed access the M0 operand its expansion in
+  // AMDGPULowerVGPREncoding expects. There are too many of these opcodes - one
+  // per register class, load and store - to list as switch cases, so they are
+  // matched by class ahead of it.
+  //
+  // Under the VGPR indexing mode the expansion emits an s_set_gpr_idx_on, which
+  // reads the index out of its SGPR and clobbers M0, so the access only has to
+  // declare that clobber. With movrel the index has to be in M0, so copy it
+  // there and rewrite the access to read M0, which lets the index computation
+  // coalesce into the copy.
+  //
+  // This runs at FinalizeISel, after SIFixSGPRCopies has made a divergent index
+  // uniform with a waterfall loop, so the copy lands inside that loop next to
+  // the per-iteration index it has to carry. Assigning M0 during selection
+  // instead would defeat the waterfall, which legalizeOperands builds only when
+  // the index is not already in an SGPR class - and M0 is one.
+  //
+  // TODO: The copy is only needed because M0 is reserved. Once it is not, the
+  // index can be an ordinary virtual register copy that the coalescer folds
+  // away, and this can go.
+  if (auto *LdSt = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
+    if (ST.useVGPRIndexMode()) {
+      if (!MI.definesRegister(AMDGPU::M0, TRI))
+        MI.addOperand(MachineOperand::CreateReg(AMDGPU::M0, /*isDef=*/true,
+                                                /*isImp=*/true));
+      return BB;
+    }
+
+    MachineOperand &IdxOp = LdSt->getIdxOp();
+    assert(IdxOp.isReg() && "VGPR-memory index must be a register");
+    if (IdxOp.getReg() != AMDGPU::M0) {
+      BuildMI(*BB, &MI, DL, TII->get(AMDGPU::COPY), AMDGPU::M0)
+          .addReg(IdxOp.getReg());
+      IdxOp.setReg(AMDGPU::M0);
+      // M0 is reserved, and the value copied into it can still be there for a
+      // later access, so this read must not kill it.
+      IdxOp.setIsKill(false);
+    }
+    return BB;
+  }
+
   switch (MI.getOpcode()) {
   case AMDGPU::WAVE_REDUCE_UMIN_PSEUDO_U32:
     return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_MIN_U32);
@@ -20474,51 +20515,6 @@ void SITargetLowering::finalizeLowering(MachineFunction &MF) const {
 
   Info->limitOccupancy(MF);
 
-  // Give a VGPR "as memory" indexed access its M0 operand, matching how
-  // AMDGPULowerVGPREncoding will expand it. Under the VGPR indexing mode the
-  // expansion emits an s_set_gpr_idx_on, which reads the index out of its SGPR
-  // and clobbers M0, so the access only has to declare that clobber. With
-  // movrel the index has to be in M0, so copy it there and rewrite the access
-  // to read M0, which lets the index computation coalesce into the copy.
-  //
-  // Doing this here rather than while selecting the access is what makes a
-  // divergent index correct: both selectors make such an index uniform with a
-  // waterfall loop - the register bank legalizer before selection,
-  // SIFixSGPRCopies after it - and this runs after both, so the copy lands
-  // inside the loop, next to the per-iteration index it has to carry.
-  //
-  // TODO: The copy is only needed because M0 is reserved. Once it is not, the
-  // index can be an ordinary virtual register copy that the coalescer folds
-  // away, and this can go.
-  for (MachineBasicBlock &MBB : MF) {
-    for (MachineInstr &MI : MBB) {
-      auto *LdSt = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI);
-      if (!LdSt)
-        continue;
-
-      // With GlobalISel this runs twice, once from InstructionSelect and again
-      // from FinalizeISel, so both forms below have to be idempotent.
-      if (ST.useVGPRIndexMode()) {
-        if (!MI.definesRegister(AMDGPU::M0, TRI))
-          MI.addOperand(MachineOperand::CreateReg(AMDGPU::M0, /*isDef=*/true,
-                                                  /*isImp=*/true));
-        continue;
-      }
-
-      MachineOperand &IdxOp = LdSt->getIdxOp();
-      assert(IdxOp.isReg() && "VGPR-memory index must be a register");
-      if (IdxOp.getReg() == AMDGPU::M0)
-        continue;
-
-      BuildMI(MBB, &MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY), AMDGPU::M0)
-          .addReg(IdxOp.getReg());
-      IdxOp.setReg(AMDGPU::M0);
-      // M0 is reserved, and the value copied into it can still be there for a
-      // later access, so this read must not kill it.
-      IdxOp.setIsKill(false);
-    }
-  }
-
   if (ST.isWave32() && !MF.empty()) {
     for (auto &MBB : MF) {
       for (auto &MI : MBB) {
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index d0bc5e09fa743..77109bec84c59 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1088,12 +1088,17 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
   // The units of $idx and $offset are in dwords.
   //
   // The index reaches the hardware through M0, and how it gets there depends on
-  // the subtarget, so SITargetLowering::finalizeLowering fills that in rather
-  // than TableGen: for v_movrel[sd] it copies the index to M0 and rewrites $idx
-  // to it, making the access a plain M0 read; under the VGPR indexing mode the
-  // s_set_gpr_idx_on it expands to clobbers M0 instead, which the access
-  // carries as an implicit def. A divergent-index access is kept inside its
-  // waterfall loop by its implicit use of EXEC (see resultDependsOnExec).
+  // the subtarget, so these carry a custom inserter rather than declaring it in
+  // TableGen: for v_movrel[sd] it copies the index to M0 and rewrites $idx to
+  // it, making the access a plain M0 read; under the VGPR indexing mode the
+  // s_set_gpr_idx_on it expands to clobbers M0 instead, which the access then
+  // carries as an implicit def.
+  //
+  // The inserter runs at FinalizeISel, which is after SIFixSGPRCopies, so a
+  // divergent index has already been made uniform by its waterfall loop and the
+  // copy lands inside that loop next to the per-iteration index. Assigning M0
+  // any earlier would defeat the waterfall: legalizeOperands decides to build it
+  // from !isSGPRReg(idx), and M0 is SGPR-class.
   //
   // Unlike SI_INDIRECT_SRC/DST and the V_INDIRECT_REG_* pseudos, the storage
   // indexed here is the whole register file rather than a tuple an operand can
@@ -1105,6 +1110,7 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
       let mayLoad = 1;
       let UseNamedOperandTable = 1;
       let hasSideEffects = 0;
+      let usesCustomInserter = 1;
   }
   def V_STORE_IDX_B#rc.Size : VPseudoInstSI <
     (outs),
@@ -1113,6 +1119,7 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
       let mayStore = 1;
       let UseNamedOperandTable = 1;
       let hasSideEffects = 0;
+      let usesCustomInserter = 1;
   }
 }
 

>From 3d53dc734ef63a67d910ec6b75ef05d09707d322 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Thu, 10 Sep 2026 17:52:12 -0400
Subject: [PATCH 23/35] Use getErrorMergeValues and stdin RUN lines per review

---
 .../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp | 47 ++++-------
 .../Target/AMDGPU/AMDGPULowerVGPREncoding.cpp | 51 +++---------
 llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp  |  5 +-
 .../AMDGPU/AMDGPURegBankLegalizeRules.cpp     |  3 +-
 .../Target/AMDGPU/AMDGPURegisterBankInfo.cpp  |  1 -
 .../AMDGPU/AMDGPUTargetTransformInfo.cpp      |  5 +-
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     | 83 ++++++-------------
 llvm/lib/Target/AMDGPU/SIISelLowering.h       |  2 -
 llvm/lib/Target/AMDGPU/SIInstrInfo.cpp        | 38 +++------
 llvm/lib/Target/AMDGPU/SIInstrInfo.td         |  3 +-
 llvm/lib/Target/AMDGPU/SIInstructions.td      | 63 ++++----------
 .../AddressSpaceVGPR/as-vgpr-addrspacecast.ll |  8 +-
 .../AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll   |  4 +-
 .../AddressSpaceVGPR/as-vgpr-divergent.ll     | 14 ++--
 .../AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll  |  4 +-
 .../as-vgpr-high-registers.ll                 |  4 +-
 .../as-vgpr-index-demanded-bits.ll            |  4 +-
 .../AddressSpaceVGPR/as-vgpr-inttoptr.ll      |  4 +-
 .../AddressSpaceVGPR/as-vgpr-optnone.ll       |  4 +-
 19 files changed, 110 insertions(+), 237 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index 043edec1376ee..e5d278b055a87 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -459,8 +459,7 @@ static bool isLoadStoreSizeLegal(const GCNSubtarget &ST,
   if (AS == AMDGPUAS::CONSTANT_ADDRESS_32BIT)
     return false;
 
-  // VGPR ("as memory") accesses are never plain-legal; they are custom-lowered
-  // to G_AMDGPU_REG_LOAD/STORE.
+  // Never plain-legal; custom-lowered to G_AMDGPU_REG_LOAD/STORE.
   if (AS == AMDGPUAS::VGPR)
     return false;
 
@@ -560,10 +559,8 @@ static bool isLoadStoreLegal(const GCNSubtarget &ST, const LegalityQuery &Query)
          !hasBufferRsrcWorkaround(Ty) && !loadStoreBitcastWorkaround(Ty);
 }
 
-// Whether the VGPR ("as memory") load/store lowering handles a MemSize-bit
-// memory access producing/consuming a ValSize-bit value. Only whole-dword
-// accesses (those with a matching V_LOAD_IDX/V_STORE_IDX pseudo) are supported
-// for now; sub-dword (8/16-bit) support lands later.
+// Whether the VGPR ("as memory") lowering handles a MemSize-bit access
+// producing a ValSize-bit value. Whole-dword only for now.
 static bool isVGPRLoadStoreSizeSupported(unsigned MemSize, unsigned ValSize) {
   return MemSize == ValSize &&
          AMDGPUMI::VLoadIdxInst::tryGetOpcodeForBitWidth(MemSize) != -1;
@@ -1691,10 +1688,8 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
     // Constant 32-bit is handled by addrspacecasting the 32-bit pointer to
     // 64-bits.
     //
-    // VGPR ("as memory") accesses are custom-lowered to the legal
-    // G_AMDGPU_REG_LOAD/STORE target instructions. Always take the custom path
-    // so an unsupported (e.g. sub-dword) access is diagnosed cleanly rather
-    // than failing to legalize.
+    // Always take the custom path, so an unsupported access is diagnosed
+    // cleanly rather than failing to legalize.
     //
     // TODO: Should generalize bitcast action into coerce, which will also cover
     // inserting addrspacecasts.
@@ -2461,9 +2456,9 @@ bool AMDGPULegalizerInfo::legalizeCustom(
 Register AMDGPULegalizerInfo::getSegmentAperture(unsigned AS,
                                                  MachineRegisterInfo &MRI,
                                                  MachineIRBuilder &B) const {
-  // See SITargetLowering::getSegmentAperture: an address space without an
-  // aperture of its own round-trips through the generic address space using the
-  // shared aperture tagged with its synthetic aperture number.
+  // See SITargetLowering::getSegmentAperture: an address space with no aperture
+  // of its own round-trips through the shared one, tagged with its synthetic
+  // aperture number.
   unsigned BaseAS = AS;
   unsigned SANum = AMDGPU::getSyntheticApertureNumber(AS);
   if (SANum != AMDGPU::SyntheticAperture::None)
@@ -3503,9 +3498,8 @@ static LLT widenToNextPowerOf2(LLT Ty) {
   return Ty.changeElementSize(PowerOf2Ceil(Ty.getSizeInBits()));
 }
 
-/// Lower a whole-dword G_LOAD / G_STORE on AMDGPUAS::VGPR into a legal
-/// G_AMDGPU_REG_LOAD / G_AMDGPU_REG_STORE indexed by the pointer's dword offset
-/// (pointer >> 2). Parallels the SelectionDAG LowerLoadStoreVGPR.
+/// Lower a whole-dword G_LOAD / G_STORE on AMDGPUAS::VGPR into
+/// G_AMDGPU_REG_LOAD / G_AMDGPU_REG_STORE. Parallels LowerLoadStoreVGPR.
 static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
   MachineIRBuilder &B = Helper.MIRBuilder;
   MachineRegisterInfo &MRI = *B.getMRI();
@@ -3517,21 +3511,13 @@ static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
 
   const LLT ValTy = MRI.getType(ValReg);
   const unsigned ValSize = ValTy.getSizeInBits();
-  // The GISel selection patterns for the indexed pseudos - and for the shift /
-  // readfirstlane that compute the index - match the extended integer LLT, so
-  // build the dword index (and the normalized register value below) with
-  // integer types rather than plain scalars.
+  // The selection patterns match the extended integer LLT, so build the index
+  // and the normalized value with integer types rather than plain scalars.
   const LLT I32 = LLT::integer(32);
 
-  // Only dword-aligned whole-dword, non-extending/non-truncating accesses are
-  // implemented. Reject anything else with a diagnostic instead of failing to
-  // legalize (sub-dword support lands in a later change).
-  //
-  // The alignment is checked here rather than in the size predicate: the index
-  // built below is the pointer shifted right by two, which discards the low two
-  // bits rather than accounting for them, so an under-aligned access would
-  // silently reach the dword containing the address instead of the bytes asked
-  // for. That is a property of how the address is formed, not of the size.
+  // Alignment is checked here rather than in the size predicate: the index is
+  // the pointer >> 2, so an under-aligned access would silently reach the
+  // containing dword. That is a property of the address, not of the size.
   if (!isVGPRLoadStoreSizeSupported(MMO.getMemoryType().getSizeInBits(),
                                     ValSize) ||
       MMO.getAlign() < Align(4)) {
@@ -3551,8 +3537,7 @@ static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
   MachineInstrBuilder Two = B.buildConstant(I32, 2);
   const MachineInstrBuilder Index = B.buildLShr(I32, PtrAsInt, Two);
 
-  // Normalize the value to i32 / <N x i32> so a selection pattern always
-  // exists (e.g. for v4i8).
+  // Normalize to i32 / <N x i32> so a selection pattern exists (e.g. v4i8).
   LLT RegTy = ValTy;
   if (ValTy.getScalarSizeInBits() != 32) {
     unsigned NumDwords = ValSize / 32;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
index c065b00e22e55..f50097b01181e 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
@@ -181,12 +181,6 @@ class AMDGPULowerVGPREncoding {
   /// Handle single \p MI. \return true if changed.
   bool runOnMachineInstr(MachineInstr &MI);
 
-  /// Lower a VGPR "as memory" (address space 13) indexed load/store pseudo
-  /// (V_LOAD_IDX_B<N> / V_STORE_IDX_B<N>) into a sequence of indexed moves over
-  /// the wave's vector registers: v_movrels_b32 / v_movreld_b32 where the
-  /// subtarget has movrel, and v_mov_b32 wrapped in s_set_gpr_idx_on/off where
-  /// it indexes with the VGPR indexing mode. This replaces the pseudo, which is
-  /// erased.
   void lowerLoadStoreIdx(MachineInstr &MI);
 
   /// Compute the mode for a single \p MI given \p Ops operands
@@ -403,10 +397,8 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
   unsigned Offset = LdSt.getOffsetOp().getImm();
   unsigned NumDwords = LdSt.getBitWidth() / 32;
 
-  // A statically out-of-range dword offset is an out-of-bounds (undefined
-  // behavior) access of the VGPR "as memory" (address space 13) region. Rather
-  // than diagnose it or emit an invalid register, mask the base into the
-  // addressable VGPR range below so the access is accepted and verifier-clean.
+  // A statically out-of-range offset is undefined behavior; mask it into the
+  // addressable range below rather than emit an invalid register.
   unsigned NumAddressableVGPRs = ST->getAddressableNumVGPRs(
       MI.getMF()->getInfo<SIMachineFunctionInfo>()->getDynamicVGPRBlockSize());
 #ifndef NDEBUG
@@ -415,12 +407,8 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
          "out of bounds VGPR 'as memory' (address space 13) access");
 #endif
 
-  // Subtargets with movrel take the index from M0, which
-  // SITargetLowering::finalizeLowering has already copied it into. The rest
-  // have no movrel and index with the VGPR indexing mode instead:
-  // s_set_gpr_idx_on enables it for one operand of the moves that follow,
-  // reading the index straight out of the SGPR holding it, so no copy is needed
-  // there.
+  // With movrel the index is in M0, put there by the custom inserter. The rest
+  // use the VGPR indexing mode, which reads it from its SGPR.
   const bool UseGPRIdxMode = ST->useVGPRIndexMode();
 
   MachineInstr *SetOn = nullptr;
@@ -444,19 +432,10 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
     Opcode =
         IsStore ? AMDGPU::V_MOVRELD_B32_as_mem : AMDGPU::V_MOVRELS_B32_as_mem;
 
-  // The dword index is (M0 + $offset). Fold $offset into the base register so
-  // each dword i reads/writes VGPR($offset + i) relative to M0. Mask the offset
-  // into the addressable range so a statically out-of-bounds offset still
-  // resolves to a valid register.
-  //
-  // A move touches VGPR($offset + i) *plus M0*, which is only known at run
-  // time, so no operand can name the register it really reads or writes.
-  // Operands that name registers as memory rather than a value are therefore
-  // marked undef: the base of every move below, and the stored value as well
-  // when that is itself undef. Liveness of the registers behind this address
-  // space is consequently not expressed here, and correctness relies on nothing
-  // else being allocated to them - which is why frontend use of the address
-  // space is documented as discouraged.
+  // A move touches VGPR($offset + i) *plus M0*, known only at run time, so no
+  // operand can name it and such operands are undef. Liveness is therefore not
+  // expressed here; correctness relies on nothing else being allocated to these
+  // registers, which is why frontend use of this address space is discouraged.
   const RegState DataFlags = IsStore
                                  ? getUndefRegState(LdSt.getDataOp().isUndef())
                                  : RegState::NoFlags;
@@ -477,9 +456,7 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
                 .addReg(Base, RegState::Undef)
                 .getInstr();
 
-    // On subtargets with more than 256 addressable VGPRs the referenced
-    // register may need high address bits; reuse the S_SET_VGPR_MSB machinery
-    // to encode them. This is a no-op on movrel-only (<=256 VGPR) subtargets.
+    // Encode high address bits above 256 addressable VGPRs; else a no-op.
     if (ST->has1024AddressableVGPRs())
       runOnMachineInstr(*Mov);
   }
@@ -690,9 +667,8 @@ bool AMDGPULowerVGPREncoding::run(MachineFunction &MF) {
   TII = ST->getInstrInfo();
   TRI = ST->getRegisterInfo();
 
-  // The S_SET_VGPR_MSB encoding is only required on subtargets with more than
-  // 256 addressable VGPRs (gfx1250). On movrel-only subtargets the pass still
-  // runs, but only to lower the VGPR "as memory" indexed load/store pseudos.
+  // S_SET_VGPR_MSB is only needed above 256 addressable VGPRs, but the pass
+  // still runs elsewhere to lower the indexed load/store pseudos.
   const bool LowerVGPRMSBs = ST->has1024AddressableVGPRs();
 
   LLVM_DEBUG(dbgs() << "*** AMDGPULowerVGPREncoding on " << MF.getName()
@@ -710,16 +686,13 @@ bool AMDGPULowerVGPREncoding::run(MachineFunction &MF) {
                       << ":\n");
 
     for (auto &MI : llvm::make_early_inc_range(MBB.instrs())) {
-      // Lower VGPR "as memory" indexed load/store pseudos on any subtarget that
-      // reaches this pass (movrel-only or gfx1250). This replaces the pseudo.
+      // Lowered on every subtarget; the MSB work below is not.
       if (isa<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
         lowerLoadStoreIdx(MI);
         Changed = true;
         continue;
       }
 
-      // The remaining work only inserts VGPR MSB encoding, which is unnecessary
-      // on movrel-only subtargets.
       if (!LowerVGPRMSBs)
         continue;
 
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
index d3401c452f2b8..dfaf67b1979ff 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
@@ -249,9 +249,8 @@ void AMDGPUMCInstLower::lower(const MachineInstr *MI, MCInst &OutMI) const {
     lowerT16FmaMixFP16(MI, OutMI);
     return;
   } else if (Opcode == AMDGPU::V_MOVRELS_B32_as_mem) {
-    // Indexed accesses of the VGPR "as memory" address space use their own
-    // opcodes because they index the register file rather than a tuple; they
-    // encode as the movrel they are named after.
+    // These have their own opcodes because they index the register file rather
+    // than a tuple, but encode as the movrel they are named after.
     Opcode = AMDGPU::V_MOVRELS_B32_e32;
   } else if (Opcode == AMDGPU::V_MOVRELD_B32_as_mem) {
     Opcode = AMDGPU::V_MOVRELD_B32_e32;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp b/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
index 2b646b66af132..38d7d8a4d7ab0 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
@@ -1393,8 +1393,7 @@ RegBankLegalizeRules::RegBankLegalizeRules(const GCNSubtarget &_ST,
             {{VgprV2S16},
              {VgprV2S16, SgprV4S32_WF, Vgpr32, Vgpr32, Sgpr32_WF}}});
 
-  // VGPR ("as memory") indexed load/store: the data is a VGPR value of any
-  // register class; the dword index is made uniform (waterfall) as an SGPR.
+  // The data is a VGPR value; the dword index is waterfalled into an SGPR.
   addRulesForGOpcs({G_AMDGPU_REG_LOAD}).Any({{BRC}, {{VgprBRC}, {Sgpr32_WF}}});
 
   addRulesForGOpcs({G_AMDGPU_REG_STORE})
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
index dec1dbed0f22e..f2bd7eee43c74 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
@@ -4498,7 +4498,6 @@ AMDGPURegisterBankInfo::getInstrMapping(const MachineInstr &MI) const {
   }
   case AMDGPU::G_AMDGPU_REG_LOAD:
   case AMDGPU::G_AMDGPU_REG_STORE: {
-    // data/result is a VGPR value; the dword index is uniform (SGPR).
     OpdsMapping[0] = getVGPROpMapping(MI.getOperand(0).getReg(), MRI, *TRI);
     OpdsMapping[1] = getSGPROpMapping(MI.getOperand(1).getReg(), MRI, *TRI);
     break;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index a8ae3a15057f6..4762a9f8b2d1d 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1208,9 +1208,8 @@ bool GCNTTIImpl::isSourceOfDivergence(const Value *V) const {
 
   // Loads from the private and flat address spaces are divergent, because
   // threads can execute the load instruction with the same inputs and get
-  // different results. The same is true of the VGPR ("as memory") address
-  // space: it is a per-lane view of the vector registers, so an access at a
-  // uniform offset still yields a per-lane (divergent) value.
+  // different results. The VGPR ("as memory") space is likewise divergent: it
+  // is a per-lane view, so even a uniform offset yields a per-lane value.
   //
   // All other loads are not divergent, because if threads issue loads with the
   // same arguments, they will always get the same result.
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index b41acf72541ee..6f31e3d234430 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -6958,26 +6958,10 @@ SITargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI,
   MachineRegisterInfo &MRI = MF->getRegInfo();
   const DebugLoc &DL = MI.getDebugLoc();
 
-  // Give a VGPR "as memory" indexed access the M0 operand its expansion in
-  // AMDGPULowerVGPREncoding expects. There are too many of these opcodes - one
-  // per register class, load and store - to list as switch cases, so they are
-  // matched by class ahead of it.
-  //
-  // Under the VGPR indexing mode the expansion emits an s_set_gpr_idx_on, which
-  // reads the index out of its SGPR and clobbers M0, so the access only has to
-  // declare that clobber. With movrel the index has to be in M0, so copy it
-  // there and rewrite the access to read M0, which lets the index computation
-  // coalesce into the copy.
-  //
-  // This runs at FinalizeISel, after SIFixSGPRCopies has made a divergent index
-  // uniform with a waterfall loop, so the copy lands inside that loop next to
-  // the per-iteration index it has to carry. Assigning M0 during selection
-  // instead would defeat the waterfall, which legalizeOperands builds only when
-  // the index is not already in an SGPR class - and M0 is one.
-  //
-  // TODO: The copy is only needed because M0 is reserved. Once it is not, the
-  // index can be an ordinary virtual register copy that the coalescer folds
-  // away, and this can go.
+  // Must run after SIFixSGPRCopies, so that a divergent index is already
+  // uniform and the copy lands inside its waterfall loop. Assigning M0 during
+  // selection would suppress that loop, which legalizeOperands builds only for
+  // an index not already in an SGPR class.
   if (auto *LdSt = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
     if (ST.useVGPRIndexMode()) {
       if (!MI.definesRegister(AMDGPU::M0, TRI))
@@ -6992,8 +6976,7 @@ SITargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI,
       BuildMI(*BB, &MI, DL, TII->get(AMDGPU::COPY), AMDGPU::M0)
           .addReg(IdxOp.getReg());
       IdxOp.setReg(AMDGPU::M0);
-      // M0 is reserved, and the value copied into it can still be there for a
-      // later access, so this read must not kill it.
+      // M0 is reserved and may still hold this value at a later access.
       IdxOp.setIsKill(false);
     }
     return BB;
@@ -9486,10 +9469,9 @@ SDValue SITargetLowering::LowerINLINEASM(SDValue Op, SelectionDAG &DAG) const {
 
 SDValue SITargetLowering::getSegmentAperture(unsigned AS, const SDLoc &DL,
                                              SelectionDAG &DAG) const {
-  // An address space that has no aperture of its own round-trips through the
-  // generic address space using a synthetic aperture: the shared aperture with
-  // the aperture number in its low bits. Dereferencing such a generic pointer
-  // is undefined behaviour; the round-trip only has to preserve the value.
+  // An address space with no aperture of its own round-trips through generic
+  // using the shared aperture tagged with its aperture number. Dereferencing
+  // such a pointer is UB; the round-trip only has to preserve the value.
   unsigned BaseAS = AS;
   unsigned SANum = AMDGPU::getSyntheticApertureNumber(AS);
   if (SANum != AMDGPU::SyntheticAperture::None)
@@ -13645,14 +13627,10 @@ static bool addressMayBeAccessedAsPrivate(const MachineMemOperand *MMO,
   return true;
 }
 
-// Lower a load or store of the VGPR ("as memory") address space (13) to a
-// REG_LOAD / REG_STORE target node. The 32-bit pointer is a byte offset into
-// the wave's view of its vector registers; the target node carries the dword
-// index (pointer >> 2). Recognizing a constant dword offset is left to the
-// selection patterns, which fold an (add index, imm) shape into the pseudo.
+// Lower a VGPR ("as memory") load or store to a REG_LOAD / REG_STORE node
+// carrying the dword index (pointer >> 2).
 //
-// TODO: sub-dword (8/16-bit) accesses are not yet supported; they are
-// diagnosed as unsupported below.
+// TODO: sub-dword (8/16-bit) accesses are diagnosed as unsupported below.
 SDValue SITargetLowering::LowerLoadStoreVGPR(SDValue Op,
                                              SelectionDAG &DAG) const {
   SDLoc DL(Op);
@@ -13660,11 +13638,8 @@ SDValue SITargetLowering::LowerLoadStoreVGPR(SDValue Op,
   EVT MemVT = MemOp->getMemoryVT();
   unsigned BitWidth = MemVT.getSizeInBits();
 
-  // Only dword-aligned whole-dword, non-extending/non-truncating accesses are
-  // implemented. Reject anything else with a diagnostic (replacing the value
-  // with poison) instead of failing instruction selection. Both callers -
-  // operation legalization and the pre-ISel combine - replace the node with
-  // this result, so the diagnostic is emitted exactly once.
+  // Both callers replace the node with this result, so the diagnostic is
+  // emitted exactly once.
   auto reportUnsupported = [&]() -> SDValue {
     const Function &F = DAG.getMachineFunction().getFunction();
     DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
@@ -13672,19 +13647,15 @@ SDValue SITargetLowering::LowerLoadStoreVGPR(SDValue Op,
         "unsupported access of VGPR 'as memory' address space (13); only "
         "dword-aligned whole-dword loads and stores are implemented",
         DL.getDebugLoc()));
-    if (isa<StoreSDNode>(MemOp))
-      return MemOp->getChain();
-    return DAG.getMergeValues(
-        {DAG.getPOISON(Op.getValueType()), MemOp->getChain()}, DL);
+    SmallVector<EVT, 2> ResultTypes(Op->values());
+    return DAG.getErrorMergeValues(ResultTypes, MemOp->getChain(), DL);
   };
 
   if (BitWidth < 32)
     return reportUnsupported();
 
-  // The dword index built below is the pointer shifted right by two, which
-  // discards the low two bits rather than accounting for them, so an
-  // under-aligned access would silently reach the dword containing the address
-  // instead of the bytes asked for.
+  // The index is the pointer >> 2, so an under-aligned access would silently
+  // reach the containing dword rather than the bytes asked for.
   if (MemOp->getAlign() < Align(4))
     return reportUnsupported();
   if (auto *Load = dyn_cast<LoadSDNode>(MemOp)) {
@@ -13704,8 +13675,7 @@ SDValue SITargetLowering::LowerLoadStoreVGPR(SDValue Op,
   SDValue Index = DAG.getNode(ISD::SRL, DL, MVT::i32, MemOp->getBasePtr(),
                               DAG.getConstant(2, DL, MVT::i32));
 
-  // View the access as i32 / <N x i32> when the memory type is not register
-  // legal (e.g. v4i8), bitcasting the value across.
+  // View as i32 / <N x i32> when the memory type is not register legal.
   EVT RegVT = MemVT;
   if (!isTypeLegal(RegVT)) {
     unsigned NumDwords = BitWidth / 32;
@@ -19415,9 +19385,8 @@ SDValue SITargetLowering::PerformDAGCombine(SDNode *N,
       return Res;
     break;
   case ISD::LOAD:
-    // Lower a VGPR ("as memory") address space (13) load to a REG_LOAD target
-    // node. Done here (not via operation legalization) so it also fires at -O0,
-    // where a scalar load is otherwise Legal and never reaches LowerLOAD.
+    // Done here rather than via operation legalization so it also fires at
+    // -O0, where a scalar load is Legal and never reaches LowerLOAD.
     if (cast<LoadSDNode>(N)->getAddressSpace() == AMDGPUAS::VGPR)
       if (SDValue V = LowerLoadStoreVGPR(SDValue(N, 0), DCI.DAG))
         return V;
@@ -20895,17 +20864,13 @@ bool SITargetLowering::isSDNodeSourceOfDivergence(const SDNode *N,
   case ISD::LOAD: {
     const LoadSDNode *L = cast<LoadSDNode>(N);
     unsigned AS = L->getAddressSpace();
-    // A flat load may access private memory. A load of the VGPR "as memory"
-    // address space reads this lane's own registers, so it is divergent however
-    // uniform the index is - and it is still an ISD::LOAD until the pre-ISel
-    // combine turns it into a REG_LOAD below.
+    // A VGPR "as memory" load reads this lane's own registers, so it is
+    // divergent however uniform the index is.
     return AS == AMDGPUAS::PRIVATE_ADDRESS || AS == AMDGPUAS::FLAT_ADDRESS ||
            AS == AMDGPUAS::VGPR;
   }
-  // The lowered form of the above. Without this the DAG takes the node's
-  // divergence to be that of its operands, so a uniform index makes the loaded
-  // value look uniform, and a consumer that requires a uniform operand gets a
-  // v_readfirstlane - broadcasting one lane's value to the whole wave.
+  // As above, after the pre-ISel combine. Without this a uniform index would
+  // make the loaded value look uniform and consumers would v_readfirstlane it.
   case AMDGPUISD::REG_LOAD:
     return true;
   case ISD::CALLSEQ_END:
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.h b/llvm/lib/Target/AMDGPU/SIISelLowering.h
index d3b9c708ff644..6986892f5e25f 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.h
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.h
@@ -128,8 +128,6 @@ class SITargetLowering final : public AMDGPUTargetLowering {
 
   SDValue widenLoad(LoadSDNode *Ld, DAGCombinerInfo &DCI) const;
   SDValue LowerLOAD(SDValue Op, SelectionDAG &DAG) const;
-  // Lower a load/store of the VGPR ("as memory") address space (13) to a
-  // REG_LOAD/REG_STORE target node indexed by the pointer's dword offset.
   SDValue LowerLoadStoreVGPR(SDValue Op, SelectionDAG &DAG) const;
   SDValue LowerSELECT(SDValue Op, SelectionDAG &DAG) const;
   SDValue lowerFastUnsafeFDIV(SDValue Op, SelectionDAG &DAG) const;
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index cb72c11403088..b2260ec72e3b2 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -308,10 +308,8 @@ bool SIInstrInfo::isSrc1DPPRevOpcode(const GCNSubtarget &ST, uint32_t Opcode) {
 bool SIInstrInfo::resultDependsOnExec(const MachineInstr &MI) const {
   assert(isVALU(MI, /*AllowLDSDMA=*/true));
 
-  // A VGPR "as memory" indexed access reads or writes the per-lane vector
-  // registers of the active lanes, so which lanes are active is part of what it
-  // does. Its implicit use of EXEC must not be treated as ignorable, or the
-  // access could be moved across a write to EXEC.
+  // Which lanes are active is part of what such an access does, so its implicit
+  // use of EXEC is not ignorable; otherwise it could move across an EXEC write.
   if (isa<AMDGPUMI::VLoadStoreIdxInst>(MI))
     return true;
 
@@ -675,13 +673,9 @@ bool SIInstrInfo::getMemOperandsWithOffsetWidth(
     BaseOp = &LdStIdx->getIdxOp();
     OffsetOp = &LdStIdx->getOffsetOp();
 
-    // Callers compare two accesses by base operand and constant offset, and
-    // treat identical bases as the same address. That only holds while the base
-    // names a value. On a movrel subtarget every one of these reads M0, so two
-    // accesses with unrelated indices compare as the same base and would be
-    // declared disjoint on their offsets alone; M0 can also be redefined
-    // between them, which the comparison does not look for. Report such an
-    // access as opaque instead.
+    // Callers treat identical base operands as the same address, which only
+    // holds while the base names a value. On a movrel subtarget these all read
+    // M0, and M0 can be redefined between them, so report them as opaque.
     if (!BaseOp->isReg() || !BaseOp->getReg().isVirtual())
       return false;
 
@@ -4331,8 +4325,7 @@ bool SIInstrInfo::areMemAccessesTriviallyDisjoint(const MachineInstr &MIa,
   if (MIa.isBundle() || MIb.isBundle())
     return false;
 
-  // VGPR "as memory" indexed accesses only alias each other, and then only
-  // when their [idx+offset, idx+offset+width) dword ranges overlap.
+  // These only alias each other, and only on overlapping dword ranges.
   const bool IsLdStIdxA = isa<AMDGPUMI::VLoadStoreIdxInst>(MIa);
   const bool IsLdStIdxB = isa<AMDGPUMI::VLoadStoreIdxInst>(MIb);
   if (IsLdStIdxA || IsLdStIdxB) {
@@ -7796,14 +7789,12 @@ SIInstrInfo::legalizeOperands(MachineInstr &MI,
     return CreatedBB;
   }
 
-  // A VGPR "as memory" indexed load/store needs its dword index in an SGPR (it
-  // becomes M0). A divergent (VGPR) index is made uniform with a waterfall
-  // loop that executes the access once per unique index across the wave.
+  // The dword index must be in an SGPR (it becomes M0), so a divergent index is
+  // made uniform with a waterfall loop.
   if (auto *LdStIdx = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
     MachineOperand *Idx = &LdStIdx->getIdxOp();
-    // Waterfall any non-SGPR index. isSGPRReg handles both virtual and physical
-    // registers, so a physical (non-SGPR) index - not expected here, but still
-    // possible - is made uniform rather than silently skipped.
+    // isSGPRReg handles physical registers too, so an unexpected physical index
+    // is waterfalled rather than silently skipped.
     if (Idx->isReg() && !RI.isSGPRReg(MRI, Idx->getReg()))
       CreatedBB = generateWaterFallLoop(*this, MI, {Idx}, MDT);
     return CreatedBB;
@@ -11326,9 +11317,8 @@ SIInstrInfo::getGenericValueUniformity(const MachineInstr &MI) const {
     return ValueUniformity::Default;
   }
 
-  // A VGPR ("as memory") indexed load is always divergent: it reads the wave's
-  // per-lane view of its vector registers, so even a uniform index yields a
-  // per-lane (divergent) value.
+  // Always divergent: it reads the wave's per-lane registers, so even a uniform
+  // index yields a per-lane value.
   if (Opcode == AMDGPU::G_AMDGPU_REG_LOAD)
     return ValueUniformity::NeverUniform;
 
@@ -11443,9 +11433,7 @@ ValueUniformity SIInstrInfo::getValueUniformity(const MachineInstr &MI) const {
     return ValueUniformity::Default;
   }
 
-  // As above for the generic opcodes, but after instruction selection: an
-  // indexed load reads the wave's per-lane view of its vector registers, so
-  // even a uniform index yields a divergent value.
+  // As above, after instruction selection.
   if (isa<AMDGPUMI::VLoadIdxInst>(MI))
     return ValueUniformity::NeverUniform;
 
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.td b/llvm/lib/Target/AMDGPU/SIInstrInfo.td
index 3a7960f6abfb0..554537e459b4c 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.td
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.td
@@ -59,8 +59,7 @@ def GFX10Gen         : GFXGen<isGFX10Only, "GFX10", "_gfx10", SIEncodingFamily.G
 // modifier behavior with dx10_enable.
 def AMDGPUclamp : SDNode<"AMDGPUISD::CLAMP", SDTFPUnaryOp>;
 
-// VGPR address space (13) load/store with a dword index operand. The index is
-// the byte offset into the wave's view of its vector registers, divided by 4.
+// VGPR address space (13) load/store indexed by (byte offset >> 2).
 def SDTRegIdxLoad : SDTypeProfile<1, 1,
     [SDTCisVT<1, i32>]>; // dword_index
 def SDTRegIdxStore : SDTypeProfile<0, 2,
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index 77109bec84c59..5f7691a105743 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1064,18 +1064,8 @@ def SI_INDIRECT_DST_V32 : SI_INDIRECT_DST<VReg_1024>;
 // VGPR "as memory" indexed load/store pseudos (address space 13)
 //===----------------------------------------------------------------------===//
 
-// V_LOAD_IDX_B<N> / V_STORE_IDX_B<N> load or store N bits from/to the wave's
-// view of its vector registers, at a dword index of ($idx + $offset). $idx is
-// a 32-bit value that may be uniform (SGPR) or divergent (VGPR); $offset is a
-// constant dword offset folded in at selection.
-//
-// AMDGPULowerVGPREncoding lowers each into an M0-relative move over the wave's
-// vector registers: v_movrels_b32 (load) / v_movreld_b32 (store) where the
-// subtarget has movrel, and a v_mov_b32 under the VGPR indexing mode otherwise.
-
-// Populates the VLdStIdxOpcodeInfo searchable table (see AMDGPUMachineInstrs.h
-// and AMDGPU::getVLdStIdxOpcodeInfo*), mapping each pseudo to its bit width and
-// load/store direction.
+// Populates the VLdStIdxOpcodeInfo searchable table, mapping each pseudo to its
+// bit width and load/store direction (see AMDGPUMachineInstrs.h).
 class VLdStIdxOpcodeInfo<int size, bit isStore> {
   Instruction Opcode = !cast<Instruction>(NAME);
   bits<12> BitWidth = size;
@@ -1085,24 +1075,12 @@ class VLdStIdxOpcodeInfo<int size, bit isStore> {
 foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
               VReg_224, VReg_256, VReg_288, VReg_320, VReg_352, VReg_384,
               VReg_512, VReg_1024] in {
-  // The units of $idx and $offset are in dwords.
-  //
-  // The index reaches the hardware through M0, and how it gets there depends on
-  // the subtarget, so these carry a custom inserter rather than declaring it in
-  // TableGen: for v_movrel[sd] it copies the index to M0 and rewrites $idx to
-  // it, making the access a plain M0 read; under the VGPR indexing mode the
-  // s_set_gpr_idx_on it expands to clobbers M0 instead, which the access then
-  // carries as an implicit def.
-  //
-  // The inserter runs at FinalizeISel, which is after SIFixSGPRCopies, so a
-  // divergent index has already been made uniform by its waterfall loop and the
-  // copy lands inside that loop next to the per-iteration index. Assigning M0
-  // any earlier would defeat the waterfall: legalizeOperands decides to build it
-  // from !isSGPRReg(idx), and M0 is SGPR-class.
+  // $idx and $offset are in dwords. The index reaches M0 by a route that
+  // depends on the subtarget, hence the custom inserter.
   //
-  // Unlike SI_INDIRECT_SRC/DST and the V_INDIRECT_REG_* pseudos, the storage
-  // indexed here is the whole register file rather than a tuple an operand can
-  // name, so these are memory accesses carrying an MMO instead.
+  // Unlike SI_INDIRECT_SRC/DST and V_INDIRECT_REG_*, the storage indexed here
+  // is the whole register file rather than a tuple an operand can name, so
+  // these are memory accesses carrying an MMO.
   def V_LOAD_IDX_B#rc.Size : VPseudoInstSI <
     (outs rc:$data),
     (ins SReg_32:$idx, i32imm:$offset)>,
@@ -1125,16 +1103,12 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
 
 // Copies of v_movrel[sd]_b32 for the moves the pseudos above expand into.
 //
-// The storage a movrel indexes is normally a register tuple, and the machine
-// verifier requires an implicit use of that tuple naming the register the
-// instruction really touches. Here the storage is the whole register file,
-// which no operand can name, so these carry their own opcodes and leave those
-// rules to the tuple form. AMDGPUMCInstLower maps them back for encoding.
-// AMDGPULowerVGPREncoding reaches $vdst and $src0 by name to decide which high
-// address bits a subtarget with more than 256 addressable VGPRs needs, so these
-// have to be in the named operand table. Without it that lookup finds nothing,
-// no S_SET_VGPR_MSB is emitted, and a base register at or above 256 is encoded
-// as its low eight bits - a silent access of the wrong register.
+// The verifier requires a movrel to implicitly use the tuple it indexes. Here
+// the storage is the whole register file, which no operand can name, so these
+// carry their own opcodes; AMDGPUMCInstLower maps them back for encoding.
+// AMDGPULowerVGPREncoding reaches $vdst and $src0 by name, so these must be in
+// the named operand table. Without it no S_SET_VGPR_MSB is emitted and a base
+// register at or above 256 is silently encoded as its low eight bits.
 let VOP1 = 1, Uses = [M0, EXEC], hasSideEffects = 0,
     UseNamedOperandTable = 1, Size = V_MOV_B32_e32.Size in {
   def V_MOVRELS_B32_as_mem
@@ -1143,10 +1117,7 @@ let VOP1 = 1, Uses = [M0, EXEC], hasSideEffects = 0,
       : VPseudoInstSI<(outs), (ins VGPR_32:$vdst, VSrc_b32:$src0)>;
 }
 
-// Select the REG_LOAD/REG_STORE target nodes into the sized indexed pseudos.
-// The pointer is a byte offset into the file; SIreg_load/SIreg_store carry the
-// dword index (ptr >> 2), and an (add idx, imm) shape folds a constant dword
-// offset into the pseudo's $offset operand.
+// An (add idx, imm) shape folds a constant dword offset into $offset.
 multiclass VRegIdxLoadStorePat<ValueType vt> {
   defvar load_inst = !cast<Instruction>("V_LOAD_IDX_B"#vt.Size);
   defvar store_inst = !cast<Instruction>("V_STORE_IDX_B"#vt.Size);
@@ -4939,10 +4910,8 @@ def G_AMDGPU_BUFFER_STORE_FORMAT_D16 : BufferStoreGenericInstruction;
 def G_AMDGPU_TBUFFER_STORE_FORMAT : TBufferStoreGenericInstruction;
 def G_AMDGPU_TBUFFER_STORE_FORMAT_D16 : TBufferStoreGenericInstruction;
 
-// GlobalISel equivalents of the REG_LOAD / REG_STORE target nodes: a load or
-// store of the VGPR ("as memory") address space, indexed by the pointer's
-// dword offset. They select to the V_LOAD_IDX_B<N> / V_STORE_IDX_B<N> pseudos
-// through the same TableGen patterns (see GINodeEquiv in AMDGPUGISel.td).
+// GlobalISel equivalents, selected through the same patterns (GINodeEquiv in
+// AMDGPUGISel.td).
 def G_AMDGPU_REG_LOAD : AMDGPUGenericInstruction {
   let OutOperandList = (outs type0:$dst);
   let InOperandList = (ins type1:$dword_index);
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll
index 05efaddedc1e1..0e4a97bf1fc31 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll
@@ -1,8 +1,8 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
-; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
-; RUN: llc -global-isel=0 -mtriple=amdgpu12.50-- -o - %s | FileCheck %s --check-prefixes=GFX1250,GFX1250-SDAG
-; RUN: llc -global-isel=1 -mtriple=amdgpu12.50-- -o - %s | FileCheck %s --check-prefixes=GFX1250,GFX1250-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.50-- < %s | FileCheck %s --check-prefixes=GFX1250,GFX1250-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.50-- < %s | FileCheck %s --check-prefixes=GFX1250,GFX1250-GISEL
 
 ; A pointer to the VGPR "as memory" address space (13) round-trips through the
 ; generic address space using a synthetic aperture: the shared aperture with the
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll
index d4c916671796c..f39bf0621cd28 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-copy.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
-; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
 
 ; Copy from VGPR "as memory" (address space 13) to global memory across the
 ; range of legal whole-dword access sizes (32 up to 1024 bits). Each uniform
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
index e779e6ff4d334..a69908541ff71 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
@@ -1,8 +1,8 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
-; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
-; RUN: llc -global-isel=0 -mtriple=amdgpu9.42-- -o - %s | FileCheck %s --check-prefixes=GFX942,GFX942-SDAG
-; RUN: llc -global-isel=1 -mtriple=amdgpu9.42-- -o - %s | FileCheck %s --check-prefixes=GFX942,GFX942-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu9.42-- < %s | FileCheck %s --check-prefixes=GFX942,GFX942-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu9.42-- < %s | FileCheck %s --check-prefixes=GFX942,GFX942-GISEL
 
 ; A divergent (per-lane) dword index into VGPR "as memory" (address space 13)
 ; is handled with a waterfall loop: for each unique index across the wave, set
@@ -33,10 +33,10 @@ define i32 @load_i32(ptr addrspace(13) %p) {
 ; GFX12-SDAG-NEXT:    s_wait_alu depctr_va_sdst(0)
 ; GFX12-SDAG-NEXT:    v_cmpx_eq_u32_e32 s2, v0
 ; GFX12-SDAG-NEXT:    s_mov_b32 m0, s2
+; GFX12-SDAG-NEXT:    ; implicit-def: $vgpr0
 ; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v1, v0
 ; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
 ; GFX12-SDAG-NEXT:    s_and_not1_wrexec_b32 s1, s1
-; GFX12-SDAG-NEXT:    ; implicit-def: $vgpr0
 ; GFX12-SDAG-NEXT:    s_cbranch_execnz .LBB0_1
 ; GFX12-SDAG-NEXT:  ; %bb.2:
 ; GFX12-SDAG-NEXT:    s_mov_b32 exec_lo, s0
@@ -135,11 +135,11 @@ define void @store_i32(ptr addrspace(13) %p, i32 %x) {
 ; GFX12-SDAG-NEXT:    s_wait_alu depctr_va_sdst(0)
 ; GFX12-SDAG-NEXT:    v_cmpx_eq_u32_e32 s2, v0
 ; GFX12-SDAG-NEXT:    s_mov_b32 m0, s2
+; GFX12-SDAG-NEXT:    ; implicit-def: $vgpr0
 ; GFX12-SDAG-NEXT:    v_movreld_b32_e32 v0, v1
+; GFX12-SDAG-NEXT:    ; implicit-def: $vgpr1
 ; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
 ; GFX12-SDAG-NEXT:    s_and_not1_wrexec_b32 s1, s1
-; GFX12-SDAG-NEXT:    ; implicit-def: $vgpr0
-; GFX12-SDAG-NEXT:    ; implicit-def: $vgpr1
 ; GFX12-SDAG-NEXT:    s_cbranch_execnz .LBB1_1
 ; GFX12-SDAG-NEXT:  ; %bb.2:
 ; GFX12-SDAG-NEXT:    s_mov_b32 exec_lo, s0
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
index ffb673f547dd4..3f40b8f10a679 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -mtriple=amdgpu9.42-- -o - %s | FileCheck %s --check-prefixes=GFX9,GFX9-SDAG
-; RUN: llc -global-isel=1 -mtriple=amdgpu9.42-- -o - %s | FileCheck %s --check-prefixes=GFX9,GFX9-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu9.42-- < %s | FileCheck %s --check-prefixes=GFX9,GFX9-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu9.42-- < %s | FileCheck %s --check-prefixes=GFX9,GFX9-GISEL
 ; RUN: llc -global-isel=0 -mtriple=amdgpu9.0a-- -filetype=null %s
 ; RUN: llc -global-isel=1 -mtriple=amdgpu9.0a-- -filetype=null %s
 
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-high-registers.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-high-registers.ll
index 75e726536fe58..be5c4d53cd31d 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-high-registers.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-high-registers.ll
@@ -1,5 +1,5 @@
-; RUN: llc -global-isel=0 -mtriple=amdgpu12.50-- -o - %s | FileCheck %s --check-prefixes=CHECK,SDAG
-; RUN: llc -global-isel=1 -mtriple=amdgpu12.50-- -o - %s | FileCheck %s --check-prefixes=CHECK,GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.50-- < %s | FileCheck %s --check-prefixes=CHECK,SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.50-- < %s | FileCheck %s --check-prefixes=CHECK,GISEL
 
 ; A subtarget with more than 256 addressable VGPRs encodes a register number's
 ; high bits separately, with S_SET_VGPR_MSB. A whole-dword access folds its
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll
index e1dfb50df7fb9..db1d8ad0bd4f6 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
-; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
 
 ; The VGPR "as memory" (address space 13) dword index only needs enough bits to
 ; address all addressable VGPRs, so a redundant high-bit mask feeding the index
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-inttoptr.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-inttoptr.ll
index ae759fa010e7a..544aebeaacbfc 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-inttoptr.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-inttoptr.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
-; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
 
 ; A constant VGPR "as memory" (address space 13) pointer formed with inttoptr:
 ; the dword index (256 >> 2 = 64) is a compile-time constant, so M0 is set from
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll
index 5d25e4522447b..cf9616322fd73 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12
-; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- -o - %s | FileCheck %s --check-prefixes=GFX12
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12
 
 ; AMDGPUAssignIdxToM0 is required lowering rather than an optimization: the
 ; v_movrel that AMDGPULowerVGPREncoding emits reads the dword index from M0, so

>From acd411044cead3cd65af3bdebf9ba26c996ba1ba Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Fri, 25 Sep 2026 10:20:43 -0400
Subject: [PATCH 24/35] Test vectors of pointers and the nonnull flag in the
 addrspacecast test

---
 .../AddressSpaceVGPR/as-vgpr-addrspacecast.ll | 179 +++++++++++++++++-
 1 file changed, 177 insertions(+), 2 deletions(-)

diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll
index 0e4a97bf1fc31..17b80a07f283a 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll
@@ -120,7 +120,7 @@ define ptr @vgpr_to_flat_nonnull(ptr addrspace(13) %ptr) {
 ; GFX1250-NEXT:    s_or_b32 s0, s1, 3
 ; GFX1250-NEXT:    v_mov_b32_e32 v1, s0
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
-  %flat = call ptr @llvm.amdgcn.addrspacecast.nonnull.p0.p13(ptr addrspace(13) %ptr)
+  %flat = addrspacecast nonnull ptr addrspace(13) %ptr to ptr
   ret ptr %flat
 }
 
@@ -139,7 +139,7 @@ define ptr addrspace(13) @flat_to_vgpr_nonnull(ptr %flat) {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
-  %v = call ptr addrspace(13) @llvm.amdgcn.addrspacecast.nonnull.p13.p0(ptr %flat)
+  %v = addrspacecast nonnull ptr %flat to ptr addrspace(13)
   ret ptr addrspace(13) %v
 }
 
@@ -162,3 +162,178 @@ define ptr addrspace(13) @vgpr_roundtrip(ptr addrspace(13) %ptr) {
   %v = addrspacecast ptr %flat to ptr addrspace(13)
   ret ptr addrspace(13) %v
 }
+
+; Vectors of pointers are cast lane by lane, each with its own null check.
+
+define <2 x ptr> @vgpr_to_flat_v2(<2 x ptr addrspace(13)> %ptr) {
+; GFX12-SDAG-LABEL: vgpr_to_flat_v2:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    v_cmp_ne_u32_e32 vcc_lo, -1, v0
+; GFX12-SDAG-NEXT:    s_mov_b64 s[0:1], src_shared_base
+; GFX12-SDAG-NEXT:    v_cmp_ne_u32_e64 s0, -1, v1
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    s_or_b32 s1, s1, 3
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0) depctr_va_vcc(0)
+; GFX12-SDAG-NEXT:    v_cndmask_b32_e64 v4, 0, s1, vcc_lo
+; GFX12-SDAG-NEXT:    v_cndmask_b32_e32 v0, 0, v0, vcc_lo
+; GFX12-SDAG-NEXT:    v_cndmask_b32_e64 v3, 0, s1, s0
+; GFX12-SDAG-NEXT:    v_cndmask_b32_e64 v2, 0, v1, s0
+; GFX12-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_4)
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v1, v4
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: vgpr_to_flat_v2:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_mov_b64 s[0:1], src_shared_base
+; GFX12-GISEL-NEXT:    v_cmp_ne_u32_e32 vcc_lo, -1, v0
+; GFX12-GISEL-NEXT:    v_cmp_ne_u32_e64 s0, -1, v1
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    s_or_b32 s1, s1, 3
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_va_vcc(0)
+; GFX12-GISEL-NEXT:    v_cndmask_b32_e32 v0, 0, v0, vcc_lo
+; GFX12-GISEL-NEXT:    v_cndmask_b32_e64 v2, 0, v1, s0
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    v_cndmask_b32_e64 v1, 0, s1, vcc_lo
+; GFX12-GISEL-NEXT:    v_cndmask_b32_e64 v3, 0, s1, s0
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-SDAG-LABEL: vgpr_to_flat_v2:
+; GFX1250-SDAG:       ; %bb.0:
+; GFX1250-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-SDAG-NEXT:    v_cmp_ne_u32_e32 vcc_lo, -1, v0
+; GFX1250-SDAG-NEXT:    s_mov_b64 s[0:1], src_shared_base
+; GFX1250-SDAG-NEXT:    v_cmp_ne_u32_e64 s0, -1, v1
+; GFX1250-SDAG-NEXT:    s_or_b32 s1, s1, 3
+; GFX1250-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
+; GFX1250-SDAG-NEXT:    v_cndmask_b32_e64 v4, 0, s1, vcc_lo
+; GFX1250-SDAG-NEXT:    v_dual_cndmask_b32 v0, 0, v0, vcc_lo :: v_dual_cndmask_b32 v2, 0, v1, s0
+; GFX1250-SDAG-NEXT:    v_cndmask_b32_e64 v3, 0, s1, s0
+; GFX1250-SDAG-NEXT:    v_mov_b32_e32 v1, v4
+; GFX1250-SDAG-NEXT:    s_set_pc_i64 s[30:31]
+;
+; GFX1250-GISEL-LABEL: vgpr_to_flat_v2:
+; GFX1250-GISEL:       ; %bb.0:
+; GFX1250-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-GISEL-NEXT:    s_mov_b64 s[0:1], src_shared_base
+; GFX1250-GISEL-NEXT:    v_cmp_ne_u32_e32 vcc_lo, -1, v0
+; GFX1250-GISEL-NEXT:    v_cmp_ne_u32_e64 s0, -1, v1
+; GFX1250-GISEL-NEXT:    s_or_b32 s1, s1, 3
+; GFX1250-GISEL-NEXT:    v_dual_cndmask_b32 v0, 0, v0, vcc_lo :: v_dual_cndmask_b32 v2, 0, v1, s0
+; GFX1250-GISEL-NEXT:    v_cndmask_b32_e64 v1, 0, s1, vcc_lo
+; GFX1250-GISEL-NEXT:    v_cndmask_b32_e64 v3, 0, s1, s0
+; GFX1250-GISEL-NEXT:    s_set_pc_i64 s[30:31]
+  %flat = addrspacecast <2 x ptr addrspace(13)> %ptr to <2 x ptr>
+  ret <2 x ptr> %flat
+}
+
+define <2 x ptr addrspace(13)> @flat_to_vgpr_v2(<2 x ptr> %flat) {
+; GFX12-LABEL: flat_to_vgpr_v2:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    v_cmp_ne_u64_e32 vcc_lo, 0, v[0:1]
+; GFX12-NEXT:    s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT:    v_cndmask_b32_e32 v0, -1, v0, vcc_lo
+; GFX12-NEXT:    v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
+; GFX12-NEXT:    s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT:    v_cndmask_b32_e32 v1, -1, v2, vcc_lo
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: flat_to_vgpr_v2:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_cmp_ne_u64_e32 vcc_lo, 0, v[0:1]
+; GFX1250-NEXT:    v_cndmask_b32_e32 v0, -1, v0, vcc_lo
+; GFX1250-NEXT:    v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
+; GFX1250-NEXT:    v_cndmask_b32_e32 v1, -1, v2, vcc_lo
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+  %v = addrspacecast <2 x ptr> %flat to <2 x ptr addrspace(13)>
+  ret <2 x ptr addrspace(13)> %v
+}
+
+define <2 x ptr> @vgpr_to_flat_nonnull_v2(<2 x ptr addrspace(13)> %ptr) {
+; GFX12-LABEL: vgpr_to_flat_nonnull_v2:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_mov_b64 s[0:1], src_shared_base
+; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT:    s_or_b32 s0, s1, 3
+; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT:    v_dual_mov_b32 v2, v1 :: v_dual_mov_b32 v1, s0
+; GFX12-NEXT:    v_mov_b32_e32 v3, s0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: vgpr_to_flat_nonnull_v2:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    s_mov_b64 s[0:1], src_shared_base
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    s_or_b32 s0, s1, 3
+; GFX1250-NEXT:    v_dual_mov_b32 v2, v1 :: v_dual_mov_b32 v1, s0
+; GFX1250-NEXT:    v_mov_b32_e32 v3, s0
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+  %flat = addrspacecast nonnull <2 x ptr addrspace(13)> %ptr to <2 x ptr>
+  ret <2 x ptr> %flat
+}
+
+define <2 x ptr addrspace(13)> @flat_to_vgpr_nonnull_v2(<2 x ptr> %flat) {
+; GFX12-LABEL: flat_to_vgpr_nonnull_v2:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    v_mov_b32_e32 v1, v2
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: flat_to_vgpr_nonnull_v2:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_mov_b32_e32 v1, v2
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+  %v = addrspacecast nonnull <2 x ptr> %flat to <2 x ptr addrspace(13)>
+  ret <2 x ptr addrspace(13)> %v
+}
+
+define <2 x ptr addrspace(13)> @vgpr_roundtrip_v2(<2 x ptr addrspace(13)> %ptr) {
+; GFX12-LABEL: vgpr_roundtrip_v2:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: vgpr_roundtrip_v2:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+  %flat = addrspacecast <2 x ptr addrspace(13)> %ptr to <2 x ptr>
+  %v = addrspacecast <2 x ptr> %flat to <2 x ptr addrspace(13)>
+  ret <2 x ptr addrspace(13)> %v
+}

>From 6d8d4f456228e26c42bbd48071b984a8fe055678 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Fri, 25 Sep 2026 10:21:28 -0400
Subject: [PATCH 25/35] Write M0 for VGPR-memory accesses at selection

---
 llvm/lib/Target/AMDGPU/AMDGPU.td              |   3 +
 .../Target/AMDGPU/AMDGPULowerVGPREncoding.cpp |  12 +-
 .../lib/Target/AMDGPU/AMDGPUMachineInstrs.cpp |  30 ++-
 llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h  |  35 ++-
 .../Target/AMDGPU/AMDGPUSearchableTables.td   |   4 +-
 llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp    |  21 ++
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     |  24 --
 llvm/lib/Target/AMDGPU/SIInstrInfo.cpp        |  63 ++++-
 llvm/lib/Target/AMDGPU/SIInstrInfo.h          |   5 +-
 llvm/lib/Target/AMDGPU/SIInstructions.td      |  96 +++++--
 .../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp    |   6 +-
 llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h |   5 +-
 .../AMDGPU/MIR/never-uniform-gmir.mir         |  15 ++
 .../AMDGPU/MIR/never-uniform.mir              |  20 ++
 .../AddressSpaceVGPR/as-vgpr-divergent-m0.mir | 252 ++++++++++++++++++
 llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp |  81 ++----
 16 files changed, 510 insertions(+), 162 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir

diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index c43e6c58f2d2e..b65682dfe136e 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -3147,6 +3147,9 @@ def HasLDSFPAtomicAddF32 : Predicate<"Subtarget->hasLDSFPAtomicAddF32()">,
 
 def NotHasAddNoCarryInsts : Predicate<"!Subtarget->hasAddNoCarryInsts()">;
 
+def UseVGPRIndexMode : Predicate<"Subtarget->useVGPRIndexMode()">;
+def NotUseVGPRIndexMode : Predicate<"!Subtarget->useVGPRIndexMode()">;
+
 def HasXNACKEnabled : Predicate<"Subtarget->isXNACKEnabled()">;
 
 def NotHasTrue16BitInsts : True16PredicateClass<"!Subtarget->hasTrue16BitInsts()">,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
index f50097b01181e..0d9417700c36d 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
@@ -393,7 +393,6 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
 
   // $data is operand 0 of both the load (def) and store (use) pseudos.
   Register Data = LdSt.getDataOp().getReg();
-  MachineOperand &IdxOp = LdSt.getIdxOp();
   unsigned Offset = LdSt.getOffsetOp().getImm();
   unsigned NumDwords = LdSt.getBitWidth() / 32;
 
@@ -407,21 +406,18 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
          "out of bounds VGPR 'as memory' (address space 13) access");
 #endif
 
-  // With movrel the index is in M0, put there by the custom inserter. The rest
-  // use the VGPR indexing mode, which reads it from its SGPR.
-  const bool UseGPRIdxMode = ST->useVGPRIndexMode();
+  // The movrel form reads its index from M0; the VGPR indexing mode form from
+  // its SGPR operand.
+  const bool UseGPRIdxMode = LdSt.isGPRIdx();
 
   MachineInstr *SetOn = nullptr;
   if (UseGPRIdxMode) {
     SetOn = BuildMI(BB, MI, DL, TII->get(AMDGPU::S_SET_GPR_IDX_ON))
-                .add(IdxOp)
+                .add(LdSt.getIdxOp())
                 .addImm(IsStore ? AMDGPU::VGPRIndexMode::DST_ENABLE
                                 : AMDGPU::VGPRIndexMode::SRC0_ENABLE)
                 .getInstr();
     SetOn->getOperand(3).setIsUndef();
-  } else {
-    assert(IdxOp.isReg() && IdxOp.getReg() == AMDGPU::M0 &&
-           "movrel index should have been copied into M0");
   }
 
   unsigned Opcode;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.cpp
index 697c63e2079d1..109a47b954b90 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.cpp
@@ -18,38 +18,46 @@
 using namespace llvm;
 using namespace AMDGPUMI;
 
-unsigned VLoadStoreIdxInst::getBitWidth() const {
+static const AMDGPU::VLdStIdxOpcodeInfo &getInfo(unsigned Opcode) {
   const AMDGPU::VLdStIdxOpcodeInfo *Info =
-      AMDGPU::getVLdStIdxOpcodeInfoByOpcode(getOpcode());
+      AMDGPU::getVLdStIdxOpcodeInfoByOpcode(Opcode);
   if (!Info)
     llvm_unreachable("unsupported V_LOAD/STORE_IDX opcode");
-  return Info->BitWidth;
+  return *Info;
+}
+
+bool VLoadStoreIdxInst::isGPRIdx() const {
+  return getInfo(getOpcode()).IsGPRIdx;
+}
+
+unsigned VLoadStoreIdxInst::getBitWidth() const {
+  return getInfo(getOpcode()).BitWidth;
 }
 
-int VLoadIdxInst::tryGetOpcodeForBitWidth(unsigned Bits) {
+int VLoadIdxInst::tryGetOpcodeForBitWidth(unsigned Bits, bool IsGPRIdx) {
   const AMDGPU::VLdStIdxOpcodeInfo *Info =
-      AMDGPU::getVLdStIdxOpcodeInfoByKey(Bits, /*IsStore=*/false);
+      AMDGPU::getVLdStIdxOpcodeInfoByKey(Bits, /*IsStore=*/false, IsGPRIdx);
   if (!Info)
     return -1;
   return Info->Opcode;
 }
 
-unsigned VLoadIdxInst::getOpcodeForBitWidth(unsigned Bits) {
-  int Opcode = tryGetOpcodeForBitWidth(Bits);
+unsigned VLoadIdxInst::getOpcodeForBitWidth(unsigned Bits, bool IsGPRIdx) {
+  int Opcode = tryGetOpcodeForBitWidth(Bits, IsGPRIdx);
   assert(Opcode != -1);
   return Opcode;
 }
 
-int VStoreIdxInst::tryGetOpcodeForBitWidth(unsigned Bits) {
+int VStoreIdxInst::tryGetOpcodeForBitWidth(unsigned Bits, bool IsGPRIdx) {
   const AMDGPU::VLdStIdxOpcodeInfo *Info =
-      AMDGPU::getVLdStIdxOpcodeInfoByKey(Bits, /*IsStore=*/true);
+      AMDGPU::getVLdStIdxOpcodeInfoByKey(Bits, /*IsStore=*/true, IsGPRIdx);
   if (!Info)
     return -1;
   return Info->Opcode;
 }
 
-unsigned VStoreIdxInst::getOpcodeForBitWidth(unsigned Bits) {
-  int Opcode = tryGetOpcodeForBitWidth(Bits);
+unsigned VStoreIdxInst::getOpcodeForBitWidth(unsigned Bits, bool IsGPRIdx) {
+  int Opcode = tryGetOpcodeForBitWidth(Bits, IsGPRIdx);
   assert(Opcode != -1);
   return Opcode;
 }
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h b/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h
index 24e24aab53152..0db3b9b0d6b83 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h
@@ -21,18 +21,29 @@ namespace llvm {
 namespace AMDGPUMI {
 
 // Wrapper for the whole-dword VGPR "as memory" (address space 13) indexed
-// load/store pseudos (V_LOAD_IDX_B<N> / V_STORE_IDX_B<N>). Operand layout:
-//   load:  (outs data), (ins idx, offset)
-//   store: (outs),      (ins data, idx, offset)
-// so data/idx/offset are always operands 0/1/2.
+// load/store pseudos. The movrel form (V_LOAD_IDX_B<N> / V_STORE_IDX_B<N>)
+// reads its index from M0; the VGPR indexing mode form
+// (V_LOAD_IDX_GPR_IDX_B<N> / V_STORE_IDX_GPR_IDX_B<N>) takes it in an SGPR:
+//   movrel:   load (outs data), (ins offset)      store (ins data, offset)
+//   gpr_idx:  load (outs data), (ins idx, offset) store (ins data, idx, offset)
 class VLoadStoreIdxInst : public MachineInstr {
 public:
+  bool isGPRIdx() const;
+
   MachineOperand &getDataOp() { return getOperand(0); }
-  MachineOperand &getIdxOp() { return getOperand(1); }
-  MachineOperand &getOffsetOp() { return getOperand(2); }
+  MachineOperand &getIdxOp() {
+    assert(isGPRIdx() && "movrel form reads its index from M0");
+    return getOperand(1);
+  }
+  MachineOperand &getOffsetOp() { return getOperand(isGPRIdx() ? 2 : 1); }
   const MachineOperand &getDataOp() const { return getOperand(0); }
-  const MachineOperand &getIdxOp() const { return getOperand(1); }
-  const MachineOperand &getOffsetOp() const { return getOperand(2); }
+  const MachineOperand &getIdxOp() const {
+    assert(isGPRIdx() && "movrel form reads its index from M0");
+    return getOperand(1);
+  }
+  const MachineOperand &getOffsetOp() const {
+    return getOperand(isGPRIdx() ? 2 : 1);
+  }
 
   unsigned getBitWidth() const;
 
@@ -43,8 +54,8 @@ class VLoadStoreIdxInst : public MachineInstr {
 
 class VLoadIdxInst : public VLoadStoreIdxInst {
 public:
-  static int tryGetOpcodeForBitWidth(unsigned Bits);
-  static unsigned getOpcodeForBitWidth(unsigned Bits);
+  static int tryGetOpcodeForBitWidth(unsigned Bits, bool IsGPRIdx = false);
+  static unsigned getOpcodeForBitWidth(unsigned Bits, bool IsGPRIdx = false);
 
   static bool classof(const MachineInstr *MI) {
     const AMDGPU::VLdStIdxOpcodeInfo *Info =
@@ -55,8 +66,8 @@ class VLoadIdxInst : public VLoadStoreIdxInst {
 
 class VStoreIdxInst : public VLoadStoreIdxInst {
 public:
-  static int tryGetOpcodeForBitWidth(unsigned Bits);
-  static unsigned getOpcodeForBitWidth(unsigned Bits);
+  static int tryGetOpcodeForBitWidth(unsigned Bits, bool IsGPRIdx = false);
+  static unsigned getOpcodeForBitWidth(unsigned Bits, bool IsGPRIdx = false);
 
   static bool classof(const MachineInstr *MI) {
     const AMDGPU::VLdStIdxOpcodeInfo *Info =
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSearchableTables.td b/llvm/lib/Target/AMDGPU/AMDGPUSearchableTables.td
index b4a069efe1b3e..9df51e02f826d 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSearchableTables.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSearchableTables.td
@@ -437,12 +437,12 @@ def AMDGPUImageDMaskIntrinsicTable : GenericTable {
 def VLdStIdxOpcodeInfoTable : GenericTable {
   let FilterClass = "VLdStIdxOpcodeInfo";
   let CppTypeName = "VLdStIdxOpcodeInfo";
-  let Fields = ["Opcode", "BitWidth", "IsStore"];
+  let Fields = ["Opcode", "BitWidth", "IsStore", "IsGPRIdx"];
   let PrimaryKey = ["Opcode"];
   let PrimaryKeyName = "getVLdStIdxOpcodeInfoByOpcodeImpl";
 }
 
 def getVLdStIdxOpcodeInfoByKeyImpl : SearchIndex {
   let Table = VLdStIdxOpcodeInfoTable;
-  let Key = ["BitWidth", "IsStore"];
+  let Key = ["BitWidth", "IsStore", "IsGPRIdx"];
 }
diff --git a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
index 964ddde0fbb01..4902cd2eb9848 100644
--- a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
@@ -67,6 +67,7 @@
 #include "SIFixSGPRCopies.h"
 #include "AMDGPU.h"
 #include "AMDGPULaneMaskUtils.h"
+#include "AMDGPUMachineInstrs.h"
 #include "GCNSubtarget.h"
 #include "llvm/CodeGen/MachineDominators.h"
 #include "llvm/InitializePasses.h"
@@ -900,6 +901,21 @@ bool SIFixSGPRCopies::tryMoveVGPRConstToSGPR(
   return true;
 }
 
+// Whether the value \p Copy writes to M0 is the index of a VGPR "as memory"
+// access.
+static bool isReadByM0IndexedAccess(const MachineInstr &Copy,
+                                    const SIRegisterInfo *TRI) {
+  MachineBasicBlock::const_iterator I(Copy), E = Copy.getParent()->end();
+  while (++I != E) {
+    auto *LdSt = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&*I);
+    if (LdSt && !LdSt->isGPRIdx())
+      return true;
+    if (I->definesRegister(AMDGPU::M0, TRI))
+      return false;
+  }
+  return false;
+}
+
 bool SIFixSGPRCopies::lowerSpecialCase(MachineInstr &MI,
                                        MachineBasicBlock::iterator &I) {
   Register DstReg = MI.getOperand(0).getReg();
@@ -911,6 +927,11 @@ bool SIFixSGPRCopies::lowerSpecialCase(MachineInstr &MI,
     // the first lane. Insert a readfirstlane and hope for the best.
     const TargetRegisterClass *SrcRC = MRI->getRegClass(SrcReg);
     if (DstReg == AMDGPU::M0 && TRI->hasVectorRegisters(SrcRC)) {
+      // A VGPR "as memory" access uses it in every lane, so moveToVALU
+      // waterfalls it instead.
+      if (isReadByM0IndexedAccess(MI, TRI))
+        return false;
+
       Register TmpReg =
           MRI->createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
 
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 6f31e3d234430..f7d8762a17069 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -6958,30 +6958,6 @@ SITargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI,
   MachineRegisterInfo &MRI = MF->getRegInfo();
   const DebugLoc &DL = MI.getDebugLoc();
 
-  // Must run after SIFixSGPRCopies, so that a divergent index is already
-  // uniform and the copy lands inside its waterfall loop. Assigning M0 during
-  // selection would suppress that loop, which legalizeOperands builds only for
-  // an index not already in an SGPR class.
-  if (auto *LdSt = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
-    if (ST.useVGPRIndexMode()) {
-      if (!MI.definesRegister(AMDGPU::M0, TRI))
-        MI.addOperand(MachineOperand::CreateReg(AMDGPU::M0, /*isDef=*/true,
-                                                /*isImp=*/true));
-      return BB;
-    }
-
-    MachineOperand &IdxOp = LdSt->getIdxOp();
-    assert(IdxOp.isReg() && "VGPR-memory index must be a register");
-    if (IdxOp.getReg() != AMDGPU::M0) {
-      BuildMI(*BB, &MI, DL, TII->get(AMDGPU::COPY), AMDGPU::M0)
-          .addReg(IdxOp.getReg());
-      IdxOp.setReg(AMDGPU::M0);
-      // M0 is reserved and may still hold this value at a later access.
-      IdxOp.setIsKill(false);
-    }
-    return BB;
-  }
-
   switch (MI.getOpcode()) {
   case AMDGPU::WAVE_REDUCE_UMIN_PSEUDO_U32:
     return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_MIN_U32);
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index b2260ec72e3b2..d4f1b0dd27b64 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -670,12 +670,13 @@ bool SIInstrInfo::getMemOperandsWithOffsetWidth(
   }
 
   if (auto *LdStIdx = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&LdSt)) {
+    // Callers treat identical base operands as the same address, which only
+    // holds while the base names a value. The movrel form reads M0, and M0 can
+    // be redefined between accesses, so report it as opaque.
+    if (!LdStIdx->isGPRIdx())
+      return false;
     BaseOp = &LdStIdx->getIdxOp();
     OffsetOp = &LdStIdx->getOffsetOp();
-
-    // Callers treat identical base operands as the same address, which only
-    // holds while the base names a value. On a movrel subtarget these all read
-    // M0, and M0 can be redefined between them, so report them as opaque.
     if (!BaseOp->isReg() || !BaseOp->getReg().isVirtual())
       return false;
 
@@ -7789,9 +7790,13 @@ SIInstrInfo::legalizeOperands(MachineInstr &MI,
     return CreatedBB;
   }
 
-  // The dword index must be in an SGPR (it becomes M0), so a divergent index is
-  // made uniform with a waterfall loop.
+  // The VGPR indexing mode form takes its dword index in an SGPR, so a
+  // divergent index is made uniform with a waterfall loop. The movrel form
+  // reads M0, whose divergent writes are waterfalled when the copy into M0 is
+  // lowered.
   if (auto *LdStIdx = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
+    if (!LdStIdx->isGPRIdx())
+      return CreatedBB;
     MachineOperand *Idx = &LdStIdx->getIdxOp();
     // isSGPRReg handles physical registers too, so an unexpected physical index
     // is waterfalled rather than silently skipped.
@@ -8215,7 +8220,7 @@ void SIInstrInfo::createWaterFallForSiCall(MachineInstr *MI,
 
 void SIInstrInfo::moveToVALU(SIInstrWorklist &Worklist,
                              MachineDominatorTree *MDT) const {
-  DenseMap<MachineInstr *, V2PhysSCopyInfo> WaterFalls;
+  MapVector<MachineInstr *, V2PhysSCopyInfo> WaterFalls;
   DenseMap<MachineInstr *, bool> V2SPhyCopiesToErase;
   while (!Worklist.empty()) {
     MachineInstr &Inst = *Worklist.top();
@@ -8238,6 +8243,9 @@ void SIInstrInfo::moveToVALU(SIInstrWorklist &Worklist,
     if (Entry.first->getOpcode() == AMDGPU::SI_CALL_ISEL)
       createWaterFallForSiCall(Entry.first, MDT, Entry.second.MOs,
                                Entry.second.SGPRs);
+    else if (isa<AMDGPUMI::VLoadStoreIdxInst>(Entry.first))
+      generateWaterFallLoop(*this, *Entry.first, Entry.second.MOs, MDT, nullptr,
+                            nullptr, Entry.second.SGPRs);
   }
 
   for (std::pair<MachineInstr *, bool> Entry : V2SPhyCopiesToErase)
@@ -8281,16 +8289,45 @@ void SIInstrInfo::createReadFirstLaneFromCopyToPhysReg(
 void SIInstrInfo::handleCopyToPhysHelper(
     SIInstrWorklist &Worklist, Register DstReg, MachineInstr &Inst,
     MachineRegisterInfo &MRI,
-    DenseMap<MachineInstr *, V2PhysSCopyInfo> &WaterFalls,
+    MapVector<MachineInstr *, V2PhysSCopyInfo> &WaterFalls,
     DenseMap<MachineInstr *, bool> &V2SPhyCopiesToErase) const {
+  Register SrcReg = Inst.getOperand(1).getReg();
+  MachineBasicBlock::iterator I = Inst.getIterator();
+  MachineBasicBlock::iterator E = Inst.getParent()->end();
   if (DstReg == AMDGPU::M0) {
-    createReadFirstLaneFromCopyToPhysReg(MRI, DstReg, Inst);
+    // A VGPR "as memory" access indexes with M0 in every lane, so like the
+    // SGPR arguments of SI_CALL_ISEL below it is waterfalled. Other readers of
+    // M0 take the first lane.
+    SmallVector<MachineOperand *, 4> IdxOps;
+    bool HasOtherReaders = false;
+    while (++I != E) {
+      auto *LdSt = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&*I);
+      if (LdSt && !LdSt->isGPRIdx())
+        IdxOps.push_back(I->findRegisterUseOperand(DstReg, &RI));
+      else if (I->readsRegister(DstReg, &RI))
+        HasOtherReaders = true;
+      if (I->findRegisterDefOperand(DstReg, &RI))
+        break;
+    }
+    // The waterfall loop reads the whole register.
+    if (!IdxOps.empty() && Inst.getOperand(1).getSubReg()) {
+      SrcReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
+      BuildMI(*Inst.getParent(), Inst, Inst.getDebugLoc(), get(AMDGPU::COPY),
+              SrcReg)
+          .addReg(Inst.getOperand(1).getReg(), {},
+                  Inst.getOperand(1).getSubReg());
+    }
+    for (MachineOperand *MO : IdxOps) {
+      MO->setReg(SrcReg);
+      V2PhysSCopyInfo &V2SCopyInfo = WaterFalls[MO->getParent()];
+      V2SCopyInfo.MOs.push_back(MO);
+      V2SCopyInfo.SGPRs.push_back(DstReg);
+    }
+    if (IdxOps.empty() || HasOtherReaders)
+      createReadFirstLaneFromCopyToPhysReg(MRI, DstReg, Inst);
     V2SPhyCopiesToErase.try_emplace(&Inst, true);
     return;
   }
-  Register SrcReg = Inst.getOperand(1).getReg();
-  MachineBasicBlock::iterator I = Inst.getIterator();
-  MachineBasicBlock::iterator E = Inst.getParent()->end();
   // Only search current block since phyreg's def & use cannot cross
   // blocks when MF.NoPhi = false.
   while (++I != E) {
@@ -8325,7 +8362,7 @@ void SIInstrInfo::handleCopyToPhysHelper(
 
 void SIInstrInfo::moveToVALUImpl(
     SIInstrWorklist &Worklist, MachineDominatorTree *MDT, MachineInstr &Inst,
-    DenseMap<MachineInstr *, V2PhysSCopyInfo> &WaterFalls,
+    MapVector<MachineInstr *, V2PhysSCopyInfo> &WaterFalls,
     DenseMap<MachineInstr *, bool> &V2SPhyCopiesToErase) const {
 
   MachineBasicBlock *MBB = Inst.getParent();
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.h b/llvm/lib/Target/AMDGPU/SIInstrInfo.h
index 48eba6b2a567d..58a25962e8414 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.h
@@ -18,6 +18,7 @@
 #include "MCTargetDesc/AMDGPUMCTargetDesc.h"
 #include "SIRegisterInfo.h"
 #include "Utils/AMDGPUBaseInfo.h"
+#include "llvm/ADT/MapVector.h"
 #include "llvm/ADT/SetVector.h"
 #include "llvm/ADT/SmallPtrSet.h"
 #include "llvm/CodeGen/TargetInstrInfo.h"
@@ -1569,7 +1570,7 @@ class SIInstrInfo final : public AMDGPUGenInstrInfo {
   void
   moveToVALUImpl(SIInstrWorklist &Worklist, MachineDominatorTree *MDT,
                  MachineInstr &Inst,
-                 DenseMap<MachineInstr *, V2PhysSCopyInfo> &WaterFalls,
+                 MapVector<MachineInstr *, V2PhysSCopyInfo> &WaterFalls,
                  DenseMap<MachineInstr *, bool> &V2SPhyCopiesToErase) const;
   /// Wrapper function for generating waterfall for instruction \p MI
   /// This function take into consideration of related pre & succ instructions
@@ -1788,7 +1789,7 @@ class SIInstrInfo final : public AMDGPUGenInstrInfo {
   void handleCopyToPhysHelper(
       SIInstrWorklist &Worklist, Register DstReg, MachineInstr &Inst,
       MachineRegisterInfo &MRI,
-      DenseMap<MachineInstr *, V2PhysSCopyInfo> &WaterFalls,
+      MapVector<MachineInstr *, V2PhysSCopyInfo> &WaterFalls,
       DenseMap<MachineInstr *, bool> &V2SPhyCopiesToErase) const;
 
   // FIXME: This should be removed
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index 5f7691a105743..26ee701495bfe 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1065,39 +1065,63 @@ def SI_INDIRECT_DST_V32 : SI_INDIRECT_DST<VReg_1024>;
 //===----------------------------------------------------------------------===//
 
 // Populates the VLdStIdxOpcodeInfo searchable table, mapping each pseudo to its
-// bit width and load/store direction (see AMDGPUMachineInstrs.h).
-class VLdStIdxOpcodeInfo<int size, bit isStore> {
+// bit width, load/store direction and indexing mechanism (see
+// AMDGPUMachineInstrs.h).
+class VLdStIdxOpcodeInfo<int size, bit isStore, bit isGPRIdx> {
   Instruction Opcode = !cast<Instruction>(NAME);
   bits<12> BitWidth = size;
   bit IsStore = isStore;
+  bit IsGPRIdx = isGPRIdx;
 }
 
 foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
               VReg_224, VReg_256, VReg_288, VReg_320, VReg_352, VReg_384,
               VReg_512, VReg_1024] in {
-  // $idx and $offset are in dwords. The index reaches M0 by a route that
-  // depends on the subtarget, hence the custom inserter.
+  // $offset and the index are in dwords. As with V_INDIRECT_REG_*_MOVREL and
+  // V_INDIRECT_REG_*_GPR_IDX, the movrel form reads the index from M0, which
+  // the selection patterns write, and the VGPR indexing mode form takes it in
+  // an SGPR and clobbers M0 through the s_set_gpr_idx_on it expands to.
   //
-  // Unlike SI_INDIRECT_SRC/DST and V_INDIRECT_REG_*, the storage indexed here
-  // is the whole register file rather than a tuple an operand can name, so
-  // these are memory accesses carrying an MMO.
+  // Unlike those pseudos, the storage indexed here is the whole register file
+  // rather than a tuple an operand can name, so these are memory accesses
+  // carrying an MMO.
   def V_LOAD_IDX_B#rc.Size : VPseudoInstSI <
     (outs rc:$data),
-    (ins SReg_32:$idx, i32imm:$offset)>,
-    VLdStIdxOpcodeInfo<rc.Size, 0> {
+    (ins i32imm:$offset)>,
+    VLdStIdxOpcodeInfo<rc.Size, 0, 0> {
       let mayLoad = 1;
       let UseNamedOperandTable = 1;
       let hasSideEffects = 0;
-      let usesCustomInserter = 1;
+      let Uses = [M0, EXEC];
   }
   def V_STORE_IDX_B#rc.Size : VPseudoInstSI <
+    (outs),
+    (ins rc:$data, i32imm:$offset)>,
+    VLdStIdxOpcodeInfo<rc.Size, 1, 0> {
+      let mayStore = 1;
+      let UseNamedOperandTable = 1;
+      let hasSideEffects = 0;
+      let Uses = [M0, EXEC];
+  }
+  def V_LOAD_IDX_GPR_IDX_B#rc.Size : VPseudoInstSI <
+    (outs rc:$data),
+    (ins SReg_32:$idx, i32imm:$offset)>,
+    VLdStIdxOpcodeInfo<rc.Size, 0, 1> {
+      let mayLoad = 1;
+      let UseNamedOperandTable = 1;
+      let hasSideEffects = 0;
+      let Uses = [M0, EXEC];
+      let Defs = [M0];
+  }
+  def V_STORE_IDX_GPR_IDX_B#rc.Size : VPseudoInstSI <
     (outs),
     (ins rc:$data, SReg_32:$idx, i32imm:$offset)>,
-    VLdStIdxOpcodeInfo<rc.Size, 1> {
+    VLdStIdxOpcodeInfo<rc.Size, 1, 1> {
       let mayStore = 1;
       let UseNamedOperandTable = 1;
       let hasSideEffects = 0;
-      let usesCustomInserter = 1;
+      let Uses = [M0, EXEC];
+      let Defs = [M0];
   }
 }
 
@@ -1117,23 +1141,45 @@ let VOP1 = 1, Uses = [M0, EXEC], hasSideEffects = 0,
       : VPseudoInstSI<(outs), (ins VGPR_32:$vdst, VSrc_b32:$src0)>;
 }
 
-// An (add idx, imm) shape folds a constant dword offset into $offset.
+// An (add idx, imm) shape folds a constant dword offset into $offset. In the
+// movrel form the index is an M0 input, so the copy into M0 is emitted by
+// selection itself; a divergent one is waterfalled by SIFixSGPRCopies.
 multiclass VRegIdxLoadStorePat<ValueType vt> {
   defvar load_inst = !cast<Instruction>("V_LOAD_IDX_B"#vt.Size);
   defvar store_inst = !cast<Instruction>("V_STORE_IDX_B"#vt.Size);
+  defvar gpr_idx_load_inst = !cast<Instruction>("V_LOAD_IDX_GPR_IDX_B"#vt.Size);
+  defvar gpr_idx_store_inst =
+      !cast<Instruction>("V_STORE_IDX_GPR_IDX_B"#vt.Size);
+
+  let OtherPredicates = [NotUseVGPRIndexMode] in {
+    def : GCNPat<
+      (vt (SIreg_load (add M0, (i32 imm:$offset)))),
+      (load_inst imm:$offset)>;
+    def : GCNPat<
+      (vt (SIreg_load M0)),
+      (load_inst 0)>;
+    def : GCNPat<
+      (SIreg_store vt:$data, (add M0, (i32 imm:$offset))),
+      (store_inst $data, imm:$offset)>;
+    def : GCNPat<
+      (SIreg_store vt:$data, M0),
+      (store_inst $data, 0)>;
+  }
 
-  def : GCNPat<
-    (vt (SIreg_load (add i32:$idx, (i32 imm:$offset)))),
-    (load_inst $idx, imm:$offset)>;
-  def : GCNPat<
-    (vt (SIreg_load i32:$idx)),
-    (load_inst $idx, 0)>;
-  def : GCNPat<
-    (SIreg_store vt:$data, (add i32:$idx, (i32 imm:$offset))),
-    (store_inst $data, $idx, imm:$offset)>;
-  def : GCNPat<
-    (SIreg_store vt:$data, i32:$idx),
-    (store_inst $data, $idx, 0)>;
+  let OtherPredicates = [UseVGPRIndexMode] in {
+    def : GCNPat<
+      (vt (SIreg_load (add i32:$idx, (i32 imm:$offset)))),
+      (gpr_idx_load_inst $idx, imm:$offset)>;
+    def : GCNPat<
+      (vt (SIreg_load i32:$idx)),
+      (gpr_idx_load_inst $idx, 0)>;
+    def : GCNPat<
+      (SIreg_store vt:$data, (add i32:$idx, (i32 imm:$offset))),
+      (gpr_idx_store_inst $data, $idx, imm:$offset)>;
+    def : GCNPat<
+      (SIreg_store vt:$data, i32:$idx),
+      (gpr_idx_store_inst $data, $idx, 0)>;
+  }
 }
 
 foreach vt = !listconcat(
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 1f9faa076c978..471dbdc9c7890 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -489,9 +489,9 @@ const VLdStIdxOpcodeInfo *getVLdStIdxOpcodeInfoByOpcode(unsigned Opc) {
   return getVLdStIdxOpcodeInfoByOpcodeImpl(Opc);
 }
 
-const VLdStIdxOpcodeInfo *getVLdStIdxOpcodeInfoByKey(uint16_t BitWidth,
-                                                     bool IsStore) {
-  return getVLdStIdxOpcodeInfoByKeyImpl(BitWidth, IsStore);
+const VLdStIdxOpcodeInfo *
+getVLdStIdxOpcodeInfoByKey(uint16_t BitWidth, bool IsStore, bool IsGPRIdx) {
+  return getVLdStIdxOpcodeInfoByKeyImpl(BitWidth, IsStore, IsGPRIdx);
 }
 
 int getMTBUFBaseOpcode(unsigned Opc) {
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index d6a6bccef5f59..8f16cadb1d996 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -489,14 +489,15 @@ struct VLdStIdxOpcodeInfo {
   unsigned Opcode;
   uint16_t BitWidth;
   bool IsStore;
+  bool IsGPRIdx;
 };
 
 LLVM_READONLY
 const VLdStIdxOpcodeInfo *getVLdStIdxOpcodeInfoByOpcode(unsigned Opc);
 
 LLVM_READONLY
-const VLdStIdxOpcodeInfo *getVLdStIdxOpcodeInfoByKey(uint16_t BitWidth,
-                                                     bool IsStore);
+const VLdStIdxOpcodeInfo *
+getVLdStIdxOpcodeInfoByKey(uint16_t BitWidth, bool IsStore, bool IsGPRIdx);
 
 LLVM_READONLY
 int getMTBUFBaseOpcode(unsigned Opc);
diff --git a/llvm/test/Analysis/UniformityAnalysis/AMDGPU/MIR/never-uniform-gmir.mir b/llvm/test/Analysis/UniformityAnalysis/AMDGPU/MIR/never-uniform-gmir.mir
index 24825209b090f..ab5dfc98294ae 100644
--- a/llvm/test/Analysis/UniformityAnalysis/AMDGPU/MIR/never-uniform-gmir.mir
+++ b/llvm/test/Analysis/UniformityAnalysis/AMDGPU/MIR/never-uniform-gmir.mir
@@ -11,3 +11,18 @@ body:             |
     %0:_(s1) = G_AMDGPU_WHOLE_WAVE_FUNC_SETUP
     G_AMDGPU_WHOLE_WAVE_FUNC_RETURN %0(s1)
 ...
+
+# VGPR "as memory" loads read each lane's own registers, so they are not
+# uniform even with a uniform index
+---
+name:            vgpr_as_memory_load
+body:             |
+  bb.0:
+    ; CHECK-LABEL: MachineUniformityInfo for function:  @vgpr_as_memory_load
+    ; CHECK: DIVERGENT: %1: %1:_(s32) = G_AMDGPU_REG_LOAD
+    ; CHECK-NOT: DIVERGENT: %2
+    %0:_(s32) = G_CONSTANT i32 0
+    %1:_(s32) = G_AMDGPU_REG_LOAD %0(s32) :: (load (s32), addrspace 13)
+    %2:_(s32) = G_ADD %0, %0
+    S_ENDPGM 0
+...
diff --git a/llvm/test/Analysis/UniformityAnalysis/AMDGPU/MIR/never-uniform.mir b/llvm/test/Analysis/UniformityAnalysis/AMDGPU/MIR/never-uniform.mir
index 399f3eddb48a8..6327e7529bbcb 100644
--- a/llvm/test/Analysis/UniformityAnalysis/AMDGPU/MIR/never-uniform.mir
+++ b/llvm/test/Analysis/UniformityAnalysis/AMDGPU/MIR/never-uniform.mir
@@ -164,3 +164,23 @@ body:             |
     %0:vgpr_32 = V_MBCNT_HI_U32_B32_e64 -1, 0, implicit $exec
     S_ENDPGM 0
 ...
+# VGPR "as memory" loads read each lane's own registers, so they are not
+# uniform even with a uniform index
+---
+name:            vgpr_as_memory_loads
+tracksRegLiveness: true
+machineFunctionInfo:
+  isEntryFunction: true
+body:             |
+  bb.0:
+    ; CHECK-LABEL: MachineUniformityInfo for function:  @vgpr_as_memory_loads
+    ; CHECK: DIVERGENT: %1
+    ; CHECK: DIVERGENT: %2
+    ; CHECK-NOT: DIVERGENT: %3
+    %0:sreg_32 = S_MOV_B32 0
+    $m0 = COPY %0
+    %1:vgpr_32 = V_LOAD_IDX_B32 0, implicit $m0, implicit $exec :: (load (s32), addrspace 13)
+    %2:vgpr_32 = V_LOAD_IDX_GPR_IDX_B32 %0, 0, implicit-def dead $m0, implicit $m0, implicit $exec :: (load (s32), addrspace 13)
+    %3:vgpr_32 = V_MOV_B32_e32 %0, implicit $exec
+    S_ENDPGM 0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir
new file mode 100644
index 0000000000000..73d691c0c7473
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir
@@ -0,0 +1,252 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6
+# RUN: llc -mtriple=amdgpu12.00-- -run-pass=si-fix-sgpr-copies -o - %s | FileCheck %s
+
+# A VGPR "as memory" access on a movrel subtarget reads its index from M0 in
+# every lane, so a divergent value copied into M0 for it is waterfalled rather
+# than taken from the first lane.
+
+---
+name:            divergent_load
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0
+
+    ; CHECK-LABEL: name: divergent_load
+    ; CHECK: successors: %bb.1(0x80000000)
+    ; CHECK-NEXT: liveins: $vgpr0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[COPY:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+    ; CHECK-NEXT: [[S_MOV_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_MOV_B32 $exec_lo
+    ; CHECK-NEXT: [[S_MOV_B32_1:%[0-9]+]]:sreg_32_xm0_xexec = S_MOV_B32 $exec_lo
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .1:
+    ; CHECK-NEXT: successors: %bb.2(0x80000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[PHI:%[0-9]+]]:sreg_32_xm0_xexec = PHI [[S_MOV_B32_1]], %bb.0, %4, %bb.2
+    ; CHECK-NEXT: [[V_READFIRSTLANE_B32_:%[0-9]+]]:sreg_32_xm0 = V_READFIRSTLANE_B32 [[COPY]], implicit $exec
+    ; CHECK-NEXT: V_CMPX_EQ_U32_nosdst_e32_term [[V_READFIRSTLANE_B32_]], [[COPY]], implicit-def $exec, implicit $exec
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .2:
+    ; CHECK-NEXT: successors: %bb.1(0x40000000), %bb.3(0x40000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_]]
+    ; CHECK-NEXT: [[V_LOAD_IDX_B32_:%[0-9]+]]:vgpr_32 = V_LOAD_IDX_B32 1, implicit killed $m0, implicit $exec :: (load (s32), addrspace 13)
+    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def $scc
+    ; CHECK-NEXT: $exec_lo = S_MOV_B32_term [[S_ANDN2_B32_]]
+    ; CHECK-NEXT: SI_WATERFALL_LOOP %bb.1, implicit $exec
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .3:
+    ; CHECK-NEXT: $exec_lo = S_MOV_B32 [[S_MOV_B32_]]
+    ; CHECK-NEXT: $vgpr0 = COPY [[V_LOAD_IDX_B32_]]
+    ; CHECK-NEXT: SI_RETURN implicit $vgpr0
+    %0:vgpr_32 = COPY $vgpr0
+    $m0 = COPY %0
+    %1:vgpr_32 = V_LOAD_IDX_B32 1, implicit $m0, implicit $exec :: (load (s32), addrspace 13)
+    $vgpr0 = COPY %1
+    SI_RETURN implicit $vgpr0
+...
+
+---
+name:            divergent_store
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+
+    ; CHECK-LABEL: name: divergent_store
+    ; CHECK: successors: %bb.1(0x80000000)
+    ; CHECK-NEXT: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[COPY:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+    ; CHECK-NEXT: [[COPY1:%[0-9]+]]:vgpr_32 = COPY $vgpr1
+    ; CHECK-NEXT: [[S_MOV_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_MOV_B32 $exec_lo
+    ; CHECK-NEXT: [[S_MOV_B32_1:%[0-9]+]]:sreg_32_xm0_xexec = S_MOV_B32 $exec_lo
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .1:
+    ; CHECK-NEXT: successors: %bb.2(0x80000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[PHI:%[0-9]+]]:sreg_32_xm0_xexec = PHI [[S_MOV_B32_1]], %bb.0, %4, %bb.2
+    ; CHECK-NEXT: [[V_READFIRSTLANE_B32_:%[0-9]+]]:sreg_32_xm0 = V_READFIRSTLANE_B32 [[COPY]], implicit $exec
+    ; CHECK-NEXT: V_CMPX_EQ_U32_nosdst_e32_term [[V_READFIRSTLANE_B32_]], [[COPY]], implicit-def $exec, implicit $exec
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .2:
+    ; CHECK-NEXT: successors: %bb.1(0x40000000), %bb.3(0x40000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_]]
+    ; CHECK-NEXT: V_STORE_IDX_B32 [[COPY1]], 2, implicit killed $m0, implicit $exec :: (store (s32), addrspace 13)
+    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def $scc
+    ; CHECK-NEXT: $exec_lo = S_MOV_B32_term [[S_ANDN2_B32_]]
+    ; CHECK-NEXT: SI_WATERFALL_LOOP %bb.1, implicit $exec
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .3:
+    ; CHECK-NEXT: $exec_lo = S_MOV_B32 [[S_MOV_B32_]]
+    ; CHECK-NEXT: SI_RETURN
+    %0:vgpr_32 = COPY $vgpr0
+    %1:vgpr_32 = COPY $vgpr1
+    $m0 = COPY %0
+    V_STORE_IDX_B32 %1, 2, implicit $m0, implicit $exec :: (store (s32), addrspace 13)
+    SI_RETURN
+...
+
+# The loop reads a whole register, so a subregister index is copied out first.
+---
+name:            divergent_subreg
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0_vgpr1
+
+    ; CHECK-LABEL: name: divergent_subreg
+    ; CHECK: successors: %bb.1(0x80000000)
+    ; CHECK-NEXT: liveins: $vgpr0_vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[COPY:%[0-9]+]]:vreg_64 = COPY $vgpr0_vgpr1
+    ; CHECK-NEXT: [[COPY1:%[0-9]+]]:vgpr_32 = COPY [[COPY]].sub1
+    ; CHECK-NEXT: [[S_MOV_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_MOV_B32 $exec_lo
+    ; CHECK-NEXT: [[S_MOV_B32_1:%[0-9]+]]:sreg_32_xm0_xexec = S_MOV_B32 $exec_lo
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .1:
+    ; CHECK-NEXT: successors: %bb.2(0x80000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[PHI:%[0-9]+]]:sreg_32_xm0_xexec = PHI [[S_MOV_B32_1]], %bb.0, %5, %bb.2
+    ; CHECK-NEXT: [[V_READFIRSTLANE_B32_:%[0-9]+]]:sreg_32_xm0 = V_READFIRSTLANE_B32 [[COPY1]], implicit $exec
+    ; CHECK-NEXT: V_CMPX_EQ_U32_nosdst_e32_term [[V_READFIRSTLANE_B32_]], [[COPY1]], implicit-def $exec, implicit $exec
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .2:
+    ; CHECK-NEXT: successors: %bb.1(0x40000000), %bb.3(0x40000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_]]
+    ; CHECK-NEXT: [[V_LOAD_IDX_B32_:%[0-9]+]]:vgpr_32 = V_LOAD_IDX_B32 0, implicit killed $m0, implicit $exec :: (load (s32), addrspace 13)
+    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def $scc
+    ; CHECK-NEXT: $exec_lo = S_MOV_B32_term [[S_ANDN2_B32_]]
+    ; CHECK-NEXT: SI_WATERFALL_LOOP %bb.1, implicit $exec
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .3:
+    ; CHECK-NEXT: $exec_lo = S_MOV_B32 [[S_MOV_B32_]]
+    ; CHECK-NEXT: $vgpr0 = COPY [[V_LOAD_IDX_B32_]]
+    ; CHECK-NEXT: SI_RETURN implicit $vgpr0
+    %0:vreg_64 = COPY $vgpr0_vgpr1
+    $m0 = COPY %0.sub1
+    %1:vgpr_32 = V_LOAD_IDX_B32 0, implicit $m0, implicit $exec :: (load (s32), addrspace 13)
+    $vgpr0 = COPY %1
+    SI_RETURN implicit $vgpr0
+...
+
+# Other readers of the same M0 value keep taking the first lane.
+---
+name:            divergent_other_reader
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0
+
+    ; CHECK-LABEL: name: divergent_other_reader
+    ; CHECK: successors: %bb.1(0x80000000)
+    ; CHECK-NEXT: liveins: $vgpr0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[COPY:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+    ; CHECK-NEXT: [[V_READFIRSTLANE_B32_:%[0-9]+]]:sreg_32_xm0 = V_READFIRSTLANE_B32 [[COPY]], implicit $exec
+    ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_]]
+    ; CHECK-NEXT: S_SENDMSG 1, implicit $exec, implicit $m0
+    ; CHECK-NEXT: [[S_MOV_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_MOV_B32 $exec_lo
+    ; CHECK-NEXT: [[S_MOV_B32_1:%[0-9]+]]:sreg_32_xm0_xexec = S_MOV_B32 $exec_lo
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .1:
+    ; CHECK-NEXT: successors: %bb.2(0x80000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[PHI:%[0-9]+]]:sreg_32_xm0_xexec = PHI [[S_MOV_B32_1]], %bb.0, %5, %bb.2
+    ; CHECK-NEXT: [[V_READFIRSTLANE_B32_1:%[0-9]+]]:sreg_32_xm0 = V_READFIRSTLANE_B32 [[COPY]], implicit $exec
+    ; CHECK-NEXT: V_CMPX_EQ_U32_nosdst_e32_term [[V_READFIRSTLANE_B32_1]], [[COPY]], implicit-def $exec, implicit $exec
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .2:
+    ; CHECK-NEXT: successors: %bb.1(0x40000000), %bb.3(0x40000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_1]]
+    ; CHECK-NEXT: [[V_LOAD_IDX_B32_:%[0-9]+]]:vgpr_32 = V_LOAD_IDX_B32 0, implicit killed $m0, implicit $exec :: (load (s32), addrspace 13)
+    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def $scc
+    ; CHECK-NEXT: $exec_lo = S_MOV_B32_term [[S_ANDN2_B32_]]
+    ; CHECK-NEXT: SI_WATERFALL_LOOP %bb.1, implicit $exec
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .3:
+    ; CHECK-NEXT: $exec_lo = S_MOV_B32 [[S_MOV_B32_]]
+    ; CHECK-NEXT: $vgpr0 = COPY [[V_LOAD_IDX_B32_]]
+    ; CHECK-NEXT: SI_RETURN implicit $vgpr0
+    %0:vgpr_32 = COPY $vgpr0
+    $m0 = COPY %0
+    S_SENDMSG 1, implicit $exec, implicit $m0
+    %1:vgpr_32 = V_LOAD_IDX_B32 0, implicit $m0, implicit $exec :: (load (s32), addrspace 13)
+    $vgpr0 = COPY %1
+    SI_RETURN implicit $vgpr0
+...
+
+---
+name:            two_divergent_loads
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+
+    ; CHECK-LABEL: name: two_divergent_loads
+    ; CHECK: successors: %bb.4(0x80000000)
+    ; CHECK-NEXT: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[COPY:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+    ; CHECK-NEXT: [[COPY1:%[0-9]+]]:vgpr_32 = COPY $vgpr1
+    ; CHECK-NEXT: [[S_MOV_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_MOV_B32 $exec_lo
+    ; CHECK-NEXT: [[S_MOV_B32_1:%[0-9]+]]:sreg_32_xm0_xexec = S_MOV_B32 $exec_lo
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .4:
+    ; CHECK-NEXT: successors: %bb.5(0x80000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[PHI:%[0-9]+]]:sreg_32_xm0_xexec = PHI [[S_MOV_B32_1]], %bb.0, %12, %bb.5
+    ; CHECK-NEXT: [[V_READFIRSTLANE_B32_:%[0-9]+]]:sreg_32_xm0 = V_READFIRSTLANE_B32 [[COPY]], implicit $exec
+    ; CHECK-NEXT: V_CMPX_EQ_U32_nosdst_e32_term [[V_READFIRSTLANE_B32_]], [[COPY]], implicit-def $exec, implicit $exec
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .5:
+    ; CHECK-NEXT: successors: %bb.4(0x40000000), %bb.6(0x40000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_]]
+    ; CHECK-NEXT: [[V_LOAD_IDX_B32_:%[0-9]+]]:vgpr_32 = V_LOAD_IDX_B32 0, implicit killed $m0, implicit $exec :: (load (s32), addrspace 13)
+    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def $scc
+    ; CHECK-NEXT: $exec_lo = S_MOV_B32_term [[S_ANDN2_B32_]]
+    ; CHECK-NEXT: SI_WATERFALL_LOOP %bb.4, implicit $exec
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .6:
+    ; CHECK-NEXT: successors: %bb.1(0x80000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $exec_lo = S_MOV_B32 [[S_MOV_B32_]]
+    ; CHECK-NEXT: [[S_MOV_B32_2:%[0-9]+]]:sreg_32_xm0_xexec = S_MOV_B32 $exec_lo
+    ; CHECK-NEXT: [[S_MOV_B32_3:%[0-9]+]]:sreg_32_xm0_xexec = S_MOV_B32 $exec_lo
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .1:
+    ; CHECK-NEXT: successors: %bb.2(0x80000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[PHI1:%[0-9]+]]:sreg_32_xm0_xexec = PHI [[S_MOV_B32_3]], %bb.6, %7, %bb.2
+    ; CHECK-NEXT: [[V_READFIRSTLANE_B32_1:%[0-9]+]]:sreg_32_xm0 = V_READFIRSTLANE_B32 [[COPY1]], implicit $exec
+    ; CHECK-NEXT: V_CMPX_EQ_U32_nosdst_e32_term [[V_READFIRSTLANE_B32_1]], [[COPY1]], implicit-def $exec, implicit $exec
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .2:
+    ; CHECK-NEXT: successors: %bb.1(0x40000000), %bb.3(0x40000000)
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_1]]
+    ; CHECK-NEXT: [[V_LOAD_IDX_B32_1:%[0-9]+]]:vgpr_32 = V_LOAD_IDX_B32 0, implicit killed $m0, implicit $exec :: (load (s32), addrspace 13)
+    ; CHECK-NEXT: [[S_ANDN2_B32_1:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI1]], $exec_lo, implicit-def $scc
+    ; CHECK-NEXT: $exec_lo = S_MOV_B32_term [[S_ANDN2_B32_1]]
+    ; CHECK-NEXT: SI_WATERFALL_LOOP %bb.1, implicit $exec
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: .3:
+    ; CHECK-NEXT: $exec_lo = S_MOV_B32 [[S_MOV_B32_2]]
+    ; CHECK-NEXT: [[V_ADD_U32_e64_:%[0-9]+]]:vgpr_32 = V_ADD_U32_e64 [[V_LOAD_IDX_B32_]], [[V_LOAD_IDX_B32_1]], 0, implicit $exec
+    ; CHECK-NEXT: $vgpr0 = COPY [[V_ADD_U32_e64_]]
+    ; CHECK-NEXT: SI_RETURN implicit $vgpr0
+    %0:vgpr_32 = COPY $vgpr0
+    %1:vgpr_32 = COPY $vgpr1
+    $m0 = COPY %0
+    %2:vgpr_32 = V_LOAD_IDX_B32 0, implicit $m0, implicit $exec :: (load (s32), addrspace 13)
+    $m0 = COPY %1
+    %3:vgpr_32 = V_LOAD_IDX_B32 0, implicit $m0, implicit $exec :: (load (s32), addrspace 13)
+    %4:vgpr_32 = V_ADD_U32_e64 %2, %3, 0, implicit $exec
+    $vgpr0 = COPY %4
+    SI_RETURN implicit $vgpr0
+...
diff --git a/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp b/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp
index a55b78a1de628..8f5b191754d40 100644
--- a/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp
+++ b/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp
@@ -8,15 +8,10 @@
 //
 // Two properties of the VGPR "as memory" (address space 13) indexed accesses
 // are asserted here rather than in a lit test, because no pass can be made to
-// observe them: these pseudos carry implicit-def $m0, so they define a physical
-// register, and MachineLICM, MachineSink and MachineCSE all decline to touch
-// them for that reason alone. Their safety today is therefore incidental, and
-// the properties below are what it would rest on if that incidental protection
-// ever went away.
-//
-// Each is paired with an ordinary VALU that must answer the other way, so a
-// change that made the query answer uniformly fails here rather than passing
-// vacuously.
+// observe them: these pseudos read M0, which is written in any function that
+// uses them, and that alone stops MachineLICM and MachineSink from moving
+// them. Their safety today is therefore incidental, and the properties below
+// are what it would rest on if that incidental protection ever went away.
 //
 //===----------------------------------------------------------------------===//
 
@@ -45,10 +40,12 @@ TEST_F(VGPRAsMemoryTest, ExecUseIsNotIgnorable) {
 name: exec_use
 body:             |
   bb.0:
-    liveins: $m0, $vgpr0
+    liveins: $m0, $sgpr0, $vgpr0
 
-    $vgpr1 = V_LOAD_IDX_B32 $m0, 0, implicit $exec :: (load (s32), addrspace 13)
-    V_STORE_IDX_B32 $vgpr0, $m0, 0, implicit $exec :: (store (s32), addrspace 13)
+    $vgpr1 = V_LOAD_IDX_B32 0, implicit $m0, implicit $exec :: (load (s32), addrspace 13)
+    V_STORE_IDX_B32 $vgpr0, 0, implicit $m0, implicit $exec :: (store (s32), addrspace 13)
+    $vgpr1 = V_LOAD_IDX_GPR_IDX_B32 $sgpr0, 0, implicit-def dead $m0, implicit $m0, implicit $exec :: (load (s32), addrspace 13)
+    V_STORE_IDX_GPR_IDX_B32 $vgpr0, $sgpr0, 0, implicit-def dead $m0, implicit $m0, implicit $exec :: (store (s32), addrspace 13)
     $vgpr2 = V_MOV_B32_e32 0, implicit $exec
     S_ENDPGM 0
 ...
@@ -60,8 +57,8 @@ body:             |
   MachineBasicBlock *MBB = MF.getBlockNumbered(0);
 
   auto ExecUseOf = [](const MachineInstr &MI) -> const MachineOperand * {
-    for (const MachineOperand &MO : MI.operands())
-      if (MO.isReg() && MO.isImplicit() && MO.getReg() == AMDGPU::EXEC)
+    for (const MachineOperand &MO : MI.implicit_operands())
+      if (MO.getReg() == AMDGPU::EXEC)
         return &MO;
     return nullptr;
   };
@@ -71,6 +68,8 @@ body:             |
     switch (MI.getOpcode()) {
     case AMDGPU::V_LOAD_IDX_B32:
     case AMDGPU::V_STORE_IDX_B32:
+    case AMDGPU::V_LOAD_IDX_GPR_IDX_B32:
+    case AMDGPU::V_STORE_IDX_GPR_IDX_B32:
       ASSERT_NE(Exec, nullptr) << "indexed access lost its implicit EXEC";
       EXPECT_FALSE(TII->isIgnorableUse(MI, MI.getOperandNo(Exec)))
           << "an indexed access may not be moved across a write to EXEC";
@@ -87,50 +86,12 @@ body:             |
   }
 }
 
-// An indexed load reads the wave's per-lane view of its vector registers, so
-// the value is divergent however the index was computed. Reporting it as
-// possibly-uniform would invite a readfirstlane, broadcasting one lane's value
-// across the wave.
-TEST_F(VGPRAsMemoryTest, IndexedLoadIsNeverUniform) {
-  StringRef MIRString = R"MIR(
-name: uniformity
-body:             |
-  bb.0:
-    liveins: $m0
-
-    $vgpr0 = V_LOAD_IDX_B32 $m0, 0, implicit $exec :: (load (s32), addrspace 13)
-    $vgpr1 = V_MOV_B32_e32 0, implicit $exec
-    S_ENDPGM 0
-...
-)MIR";
-
-  ASSERT_TRUE(parseMIR(MIRString));
-  MachineFunction &MF = getMF("uniformity");
-  const SIInstrInfo *TII = MF.getSubtarget<GCNSubtarget>().getInstrInfo();
-  MachineBasicBlock *MBB = MF.getBlockNumbered(0);
-
-  for (MachineInstr &MI : *MBB) {
-    switch (MI.getOpcode()) {
-    case AMDGPU::V_LOAD_IDX_B32:
-      EXPECT_EQ(TII->getValueUniformity(MI), ValueUniformity::NeverUniform);
-      break;
-    case AMDGPU::V_MOV_B32_e32:
-      // The contrast: an ordinary move is only as divergent as its operands.
-      EXPECT_EQ(TII->getValueUniformity(MI), ValueUniformity::Default);
-      break;
-    default:
-      break;
-    }
-  }
-}
-
-// Two whole-dword accesses whose index operand is M0 - the form every access
-// has on a movrel subtarget - with M0 redefined between them. The dwords they
-// touch are (first M0)+1 and (second M0)+0, which are the same dword whenever
-// the second index is one more than the first, so they may alias. Disjointness
-// is decided from the base operand and the constant offset, and both bases are
-// literally $m0, so nothing in that comparison can tell the two M0 values
-// apart.
+// Two whole-dword accesses indexed by M0 - the form every access has on a
+// movrel subtarget - with M0 redefined between them. The dwords they touch are
+// (first M0)+1 and (second M0)+0, which are the same dword whenever the second
+// index is one more than the first, so they may alias. Disjointness is decided
+// from a base operand and the constant offset, and these have no base operand
+// that could tell the two M0 values apart.
 TEST_F(VGPRAsMemoryTest, M0IndexedAccessesAcrossAM0RedefMayAlias) {
   StringRef MIRString = R"MIR(
 name: m0_redef
@@ -139,9 +100,9 @@ body:             |
     liveins: $sgpr0, $sgpr1, $vgpr0
 
     $m0 = COPY $sgpr0
-    $vgpr1 = V_LOAD_IDX_B32 $m0, 1, implicit $exec :: (load (s32), addrspace 13)
+    $vgpr1 = V_LOAD_IDX_B32 1, implicit $m0, implicit $exec :: (load (s32), addrspace 13)
     $m0 = COPY $sgpr1
-    V_STORE_IDX_B32 $vgpr0, $m0, 0, implicit $exec :: (store (s32), addrspace 13)
+    V_STORE_IDX_B32 $vgpr0, 0, implicit $m0, implicit $exec :: (store (s32), addrspace 13)
     S_ENDPGM 0
 ...
 )MIR";

>From d85c0c1fa6978a887cd4684d350a1d8468ff6e78 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Fri, 25 Sep 2026 10:21:43 -0400
Subject: [PATCH 26/35] Expand atomics on the VGPR address space into plain
 accesses

---
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     |  12 ++
 .../AMDGPU/AddressSpaceVGPR/as-vgpr-atomic.ll | 120 ++++++++++++++++++
 2 files changed, 132 insertions(+)
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-atomic.ll

diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index f7d8762a17069..78daff21baefa 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -21060,6 +21060,11 @@ SITargetLowering::shouldExpandAtomicRMWInIR(const AtomicRMWInst *RMW) const {
   if (AS == AMDGPUAS::PRIVATE_ADDRESS)
     return getPrivateAtomicExpansionKind(*getSubtarget());
 
+  // Only the executing lane can access its view of the VGPRs, so as with
+  // private memory there is nothing to be atomic with respect to.
+  if (AS == AMDGPUAS::VGPR)
+    return AtomicExpansionKind::NotAtomic;
+
   // 64-bit flat atomics that dynamically reside in private memory will silently
   // be dropped.
   //
@@ -21344,6 +21349,8 @@ SITargetLowering::shouldExpandAtomicRMWInIR(const AtomicRMWInst *RMW) const {
 
 TargetLowering::AtomicExpansionKind
 SITargetLowering::shouldExpandAtomicLoadInIR(LoadInst *LI) const {
+  if (LI->getPointerAddressSpace() == AMDGPUAS::VGPR)
+    return AtomicExpansionKind::NotAtomic;
   return LI->getPointerAddressSpace() == AMDGPUAS::PRIVATE_ADDRESS
              ? getPrivateAtomicExpansionKind(*getSubtarget())
              : AtomicExpansionKind::None;
@@ -21351,6 +21358,8 @@ SITargetLowering::shouldExpandAtomicLoadInIR(LoadInst *LI) const {
 
 TargetLowering::AtomicExpansionKind
 SITargetLowering::shouldExpandAtomicStoreInIR(StoreInst *SI) const {
+  if (SI->getPointerAddressSpace() == AMDGPUAS::VGPR)
+    return AtomicExpansionKind::NotAtomic;
   return SI->getPointerAddressSpace() == AMDGPUAS::PRIVATE_ADDRESS
              ? getPrivateAtomicExpansionKind(*getSubtarget())
              : AtomicExpansionKind::None;
@@ -21363,6 +21372,9 @@ SITargetLowering::shouldExpandAtomicCmpXchgInIR(
   if (AddrSpace == AMDGPUAS::PRIVATE_ADDRESS)
     return getPrivateAtomicExpansionKind(*getSubtarget());
 
+  if (AddrSpace == AMDGPUAS::VGPR)
+    return AtomicExpansionKind::NotAtomic;
+
   if (AddrSpace != AMDGPUAS::FLAT_ADDRESS || !flatInstrMayAccessPrivate(CmpX))
     return AtomicExpansionKind::None;
 
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-atomic.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-atomic.ll
new file mode 100644
index 0000000000000..9f9d206e4879a
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-atomic.ll
@@ -0,0 +1,120 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+
+; Only the executing lane can access its view of the VGPRs, so, as for the
+; private address space, atomics on the VGPR "as memory" address space (13) are
+; performed as ordinary loads and stores.
+
+define i32 @atomicrmw_add(ptr addrspace(13) inreg %p, i32 %v) {
+; GFX12-LABEL: atomicrmw_add:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v1, v0
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT:    v_add_nc_u32_e32 v0, v1, v0
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    v_mov_b32_e32 v0, v1
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %r = atomicrmw add ptr addrspace(13) %p, i32 %v seq_cst
+  ret i32 %r
+}
+
+define float @atomicrmw_fadd(ptr addrspace(13) inreg %p, float %v) {
+; GFX12-LABEL: atomicrmw_fadd:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v1, v0
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT:    v_add_f32_e32 v0, v1, v0
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    v_mov_b32_e32 v0, v1
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %r = atomicrmw fadd ptr addrspace(13) %p, float %v monotonic
+  ret float %r
+}
+
+define i64 @atomicrmw_xchg_i64(ptr addrspace(13) inreg %p, i64 %v) {
+; GFX12-LABEL: atomicrmw_xchg_i64:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v2, v0
+; GFX12-NEXT:    v_movrels_b32_e32 v3, v1
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    v_movreld_b32_e32 v1, v1
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_3)
+; GFX12-NEXT:    v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %r = atomicrmw xchg ptr addrspace(13) %p, i64 %v seq_cst, align 8
+  ret i64 %r
+}
+
+define i32 @cmpxchg(ptr addrspace(13) inreg %p, i32 %cmp, i32 %new) {
+; GFX12-LABEL: cmpxchg:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v2, v0
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX12-NEXT:    v_cmp_eq_u32_e32 vcc_lo, v2, v0
+; GFX12-NEXT:    s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT:    v_cndmask_b32_e32 v0, v2, v1, vcc_lo
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    v_mov_b32_e32 v0, v2
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %pair = cmpxchg ptr addrspace(13) %p, i32 %cmp, i32 %new seq_cst seq_cst
+  %r = extractvalue { i32, i1 } %pair, 0
+  ret i32 %r
+}
+
+define i32 @load_atomic(ptr addrspace(13) inreg %p) {
+; GFX12-LABEL: load_atomic:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %r = load atomic i32, ptr addrspace(13) %p acquire, align 4
+  ret i32 %r
+}
+
+define void @store_atomic(ptr addrspace(13) inreg %p, i32 %v) {
+; GFX12-LABEL: store_atomic:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  store atomic i32 %v, ptr addrspace(13) %p release, align 4
+  ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX12-GISEL: {{.*}}
+; GFX12-SDAG: {{.*}}

>From 4fd951b33762af21360b4619ff167e21e9f3bfba Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Fri, 25 Sep 2026 10:21:53 -0400
Subject: [PATCH 27/35] Leave upstream code in SIInstrInfo.cpp as it was

---
 llvm/lib/Target/AMDGPU/SIInstrInfo.cpp | 11 ++++++-----
 1 file changed, 6 insertions(+), 5 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index d4f1b0dd27b64..497db750b3bf2 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -4323,6 +4323,9 @@ bool SIInstrInfo::areMemAccessesTriviallyDisjoint(const MachineInstr &MIa,
   if (MIa.hasOrderedMemoryRef() || MIb.hasOrderedMemoryRef())
     return false;
 
+  if (isLDSDMA(MIa) || isLDSDMA(MIb))
+    return false;
+
   if (MIa.isBundle() || MIb.isBundle())
     return false;
 
@@ -4335,9 +4338,6 @@ bool SIInstrInfo::areMemAccessesTriviallyDisjoint(const MachineInstr &MIa,
     return true;
   }
 
-  if (isLDSDMA(MIa) || isLDSDMA(MIb))
-    return false;
-
   // TODO: Should we check the address space from the MachineMemOperand? That
   // would allow us to distinguish objects we know don't alias based on the
   // underlying address space, even if it was lowered to a different one,
@@ -5920,6 +5920,7 @@ bool SIInstrInfo::verifyInstruction(const MachineInstr &MI,
       Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64) {
     const bool IsDst = Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e32 ||
                        Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64;
+
     const unsigned StaticNumOps =
         Desc.getNumOperands() + Desc.implicit_uses().size();
     const unsigned NumImplicitOps = IsDst ? 2 : 1;
@@ -5949,8 +5950,8 @@ bool SIInstrInfo::verifyInstruction(const MachineInstr &MI,
     }
 
     const MachineOperand &Src0 = MI.getOperand(Src0Idx);
-    const MachineOperand &ImpUse =
-        MI.getOperand(StaticNumOps + NumImplicitOps - 1);
+    const MachineOperand &ImpUse
+      = MI.getOperand(StaticNumOps + NumImplicitOps - 1);
     if (!ImpUse.isReg() || !ImpUse.isUse() ||
         !isSubRegOf(RI, ImpUse, IsDst ? *Dst : Src0)) {
       ErrInfo = "src0 should be subreg of implicit vector use";

>From 16d5af64e91ef59572e58364eae63d41dbec5c50 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Fri, 25 Sep 2026 10:22:04 -0400
Subject: [PATCH 28/35] Document the VGPR synthetic aperture

---
 llvm/docs/AMDGPUUsage.rst | 10 +++++-----
 1 file changed, 5 insertions(+), 5 deletions(-)

diff --git a/llvm/docs/AMDGPUUsage.rst b/llvm/docs/AMDGPUUsage.rst
index b41f1af00c5b7..5f82b61e46bef 100644
--- a/llvm/docs/AMDGPUUsage.rst
+++ b/llvm/docs/AMDGPUUsage.rst
@@ -1179,11 +1179,6 @@ supported for the ``amdgcn`` target.
   aligned to 2^32 which makes it easier to convert from flat to segment or
   segment to flat.
 
-  *Synthetic apertures* are defined that enable safe roundtrips of pointers
-  from special address spaces through the generic address space. Attempting to
-  dereference generic pointers obtained in this way (using e.g. ``load`` or
-  ``store``) has undefined behavior.
-
   A global address space address has the same value when used as a flat address
   so no conversion is needed.
 
@@ -1385,6 +1380,10 @@ supported for the ``amdgcn`` target.
   ``alloca`` is not visible while in a called function. Attempting to dereference
   a pointer to such memory in a called function is undefined behavior.
 
+  Pointers can be cast to and from the generic address space, which uses a
+  :ref:`synthetic aperture<amdgpu-synthetic-apertures>`; a generic pointer
+  obtained this way cannot be dereferenced.
+
 **Barrier**
   This address space represents barrier IDs (introduced in GFX12) as addresses.
   It does not map directly to any addressable memory and is implemented using
@@ -1434,6 +1433,7 @@ The following synthetic apertures are defined:
     Name         Number  Mask           Corresponding :ref:`Address Space<amdgpu-address-spaces-table>`
     ============ ======= ============== ================================================================
     BARRIER      1       ``0x00000001`` Barrier
+    VGPR         3       ``0x00000003`` VGPR
     ============ ======= ============== ================================================================
 
 Converting a pointer to generic (64 bits) using synthetic apertures is done as follows:

>From e62879a052984e1d3f03b6109b082b5d51dfb8f7 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Fri, 25 Sep 2026 10:22:15 -0400
Subject: [PATCH 29/35] Read test input from stdin in the remaining RUN lines

---
 .../CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll   | 4 ++--
 .../CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll    | 4 ++--
 2 files changed, 4 insertions(+), 4 deletions(-)

diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
index 3f40b8f10a679..25b865f5478ca 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
@@ -1,8 +1,8 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
 ; RUN: llc -global-isel=0 -mtriple=amdgpu9.42-- < %s | FileCheck %s --check-prefixes=GFX9,GFX9-SDAG
 ; RUN: llc -global-isel=1 -mtriple=amdgpu9.42-- < %s | FileCheck %s --check-prefixes=GFX9,GFX9-GISEL
-; RUN: llc -global-isel=0 -mtriple=amdgpu9.0a-- -filetype=null %s
-; RUN: llc -global-isel=1 -mtriple=amdgpu9.0a-- -filetype=null %s
+; RUN: llc -global-isel=0 -mtriple=amdgpu9.0a-- -filetype=null < %s
+; RUN: llc -global-isel=1 -mtriple=amdgpu9.0a-- -filetype=null < %s
 
 ; The VGPR "as memory" address space (13) on a subtarget that has no movrel.
 ; gfx9 indexes with the VGPR indexing mode instead, so AMDGPULowerVGPREncoding
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
index 0cbda8e8ff646..52fd329aa1428 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
@@ -1,5 +1,5 @@
-; RUN: not llc -global-isel=0 -mtriple=amdgpu12.00-- -filetype=null %s 2>&1 | FileCheck %s
-; RUN: not llc -global-isel=1 -mtriple=amdgpu12.00-- -filetype=null %s 2>&1 | FileCheck %s
+; RUN: not llc -global-isel=0 -mtriple=amdgpu12.00-- -filetype=null < %s 2>&1 | FileCheck %s
+; RUN: not llc -global-isel=1 -mtriple=amdgpu12.00-- -filetype=null < %s 2>&1 | FileCheck %s
 
 ; Accesses of the VGPR "as memory" address space (13) that are not implemented
 ; must be rejected with a clean diagnostic on both SelectionDAG and GlobalISel,

>From 2c7cf1c260e1ca108d906e77ce26291ffa6c3e63 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Fri, 25 Sep 2026 10:53:35 -0400
Subject: [PATCH 30/35] Lower extending whole-dword loads from the VGPR address
 space

---
 .../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp | 17 ++-----
 .../Target/AMDGPU/AMDGPULowerVGPREncoding.cpp | 11 +----
 llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h  |  6 +--
 .../AMDGPU/AMDGPURegBankLegalizeRules.cpp     |  1 -
 .../Target/AMDGPU/AMDGPURegisterBankInfo.cpp  |  2 -
 .../Target/AMDGPU/AMDGPUSearchableTables.td   |  4 --
 .../AMDGPU/AMDGPUTargetTransformInfo.cpp      |  4 +-
 llvm/lib/Target/AMDGPU/SIISelLowering.cpp     | 33 +++++++------
 llvm/lib/Target/AMDGPU/SIInstrInfo.cpp        | 27 +++++------
 llvm/lib/Target/AMDGPU/SIInstructions.td      | 31 ++++---------
 .../AddressSpaceVGPR/as-vgpr-addrspacecast.ll |  7 ++-
 .../AMDGPU/AddressSpaceVGPR/as-vgpr-atomic.ll |  5 +-
 .../AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll  | 46 ++++++++++++++++---
 .../AddressSpaceVGPR/as-vgpr-divergent-m0.mir |  5 +-
 .../AddressSpaceVGPR/as-vgpr-divergent.ll     | 20 +++-----
 .../AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll  | 10 +---
 .../as-vgpr-high-registers.ll                 | 18 ++------
 .../as-vgpr-index-demanded-bits.ll            |  8 ++--
 .../AddressSpaceVGPR/as-vgpr-optnone.ll       |  7 +--
 .../AddressSpaceVGPR/as-vgpr-unsupported.ll   | 19 +++-----
 llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp | 23 +++-------
 21 files changed, 124 insertions(+), 180 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index e5d278b055a87..301618170276f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -559,8 +559,7 @@ static bool isLoadStoreLegal(const GCNSubtarget &ST, const LegalityQuery &Query)
          !hasBufferRsrcWorkaround(Ty) && !loadStoreBitcastWorkaround(Ty);
 }
 
-// Whether the VGPR ("as memory") lowering handles a MemSize-bit access
-// producing a ValSize-bit value. Whole-dword only for now.
+// Whole-dword accesses only, for now.
 static bool isVGPRLoadStoreSizeSupported(unsigned MemSize, unsigned ValSize) {
   return MemSize == ValSize &&
          AMDGPUMI::VLoadIdxInst::tryGetOpcodeForBitWidth(MemSize) != -1;
@@ -1688,8 +1687,7 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
     // Constant 32-bit is handled by addrspacecasting the 32-bit pointer to
     // 64-bits.
     //
-    // Always take the custom path, so an unsupported access is diagnosed
-    // cleanly rather than failing to legalize.
+    // VGPR accesses are always custom, to be lowered or diagnosed.
     //
     // TODO: Should generalize bitcast action into coerce, which will also cover
     // inserting addrspacecasts.
@@ -2456,9 +2454,6 @@ bool AMDGPULegalizerInfo::legalizeCustom(
 Register AMDGPULegalizerInfo::getSegmentAperture(unsigned AS,
                                                  MachineRegisterInfo &MRI,
                                                  MachineIRBuilder &B) const {
-  // See SITargetLowering::getSegmentAperture: an address space with no aperture
-  // of its own round-trips through the shared one, tagged with its synthetic
-  // aperture number.
   unsigned BaseAS = AS;
   unsigned SANum = AMDGPU::getSyntheticApertureNumber(AS);
   if (SANum != AMDGPU::SyntheticAperture::None)
@@ -3511,13 +3506,11 @@ static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
 
   const LLT ValTy = MRI.getType(ValReg);
   const unsigned ValSize = ValTy.getSizeInBits();
-  // The selection patterns match the extended integer LLT, so build the index
-  // and the normalized value with integer types rather than plain scalars.
+  // The selection patterns match integer LLTs rather than plain scalars.
   const LLT I32 = LLT::integer(32);
 
-  // Alignment is checked here rather than in the size predicate: the index is
-  // the pointer >> 2, so an under-aligned access would silently reach the
-  // containing dword. That is a property of the address, not of the size.
+  // The index is the pointer >> 2, so an under-aligned access would silently
+  // reach the containing dword.
   if (!isVGPRLoadStoreSizeSupported(MMO.getMemoryType().getSizeInBits(),
                                     ValSize) ||
       MMO.getAlign() < Align(4)) {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
index 0d9417700c36d..cef26a266c60c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
@@ -391,7 +391,6 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
   const DebugLoc &DL = MI.getDebugLoc();
   const bool IsStore = LdSt.mayStore();
 
-  // $data is operand 0 of both the load (def) and store (use) pseudos.
   Register Data = LdSt.getDataOp().getReg();
   unsigned Offset = LdSt.getOffsetOp().getImm();
   unsigned NumDwords = LdSt.getBitWidth() / 32;
@@ -406,8 +405,6 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
          "out of bounds VGPR 'as memory' (address space 13) access");
 #endif
 
-  // The movrel form reads its index from M0; the VGPR indexing mode form from
-  // its SGPR operand.
   const bool UseGPRIdxMode = LdSt.isGPRIdx();
 
   MachineInstr *SetOn = nullptr;
@@ -428,10 +425,8 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
     Opcode =
         IsStore ? AMDGPU::V_MOVRELD_B32_as_mem : AMDGPU::V_MOVRELS_B32_as_mem;
 
-  // A move touches VGPR($offset + i) *plus M0*, known only at run time, so no
-  // operand can name it and such operands are undef. Liveness is therefore not
-  // expressed here; correctness relies on nothing else being allocated to these
-  // registers, which is why frontend use of this address space is discouraged.
+  // A move touches its base plus the run-time index, which no operand can name,
+  // so the base is undef and the liveness of those registers is not expressed.
   const RegState DataFlags = IsStore
                                  ? getUndefRegState(LdSt.getDataOp().isUndef())
                                  : RegState::NoFlags;
@@ -663,8 +658,6 @@ bool AMDGPULowerVGPREncoding::run(MachineFunction &MF) {
   TII = ST->getInstrInfo();
   TRI = ST->getRegisterInfo();
 
-  // S_SET_VGPR_MSB is only needed above 256 addressable VGPRs, but the pass
-  // still runs elsewhere to lower the indexed load/store pseudos.
   const bool LowerVGPRMSBs = ST->has1024AddressableVGPRs();
 
   LLVM_DEBUG(dbgs() << "*** AMDGPULowerVGPREncoding on " << MF.getName()
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h b/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h
index 0db3b9b0d6b83..600d7d6a618e1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMachineInstrs.h
@@ -20,10 +20,8 @@
 namespace llvm {
 namespace AMDGPUMI {
 
-// Wrapper for the whole-dword VGPR "as memory" (address space 13) indexed
-// load/store pseudos. The movrel form (V_LOAD_IDX_B<N> / V_STORE_IDX_B<N>)
-// reads its index from M0; the VGPR indexing mode form
-// (V_LOAD_IDX_GPR_IDX_B<N> / V_STORE_IDX_GPR_IDX_B<N>) takes it in an SGPR:
+// Wrapper for the whole-dword VGPR "as memory" (address space 13) load/store
+// pseudos. The movrel form reads its index from M0, the GPR_IDX form from $idx:
 //   movrel:   load (outs data), (ins offset)      store (ins data, offset)
 //   gpr_idx:  load (outs data), (ins idx, offset) store (ins data, idx, offset)
 class VLoadStoreIdxInst : public MachineInstr {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp b/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
index 38d7d8a4d7ab0..2369d0ae042f5 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
@@ -1393,7 +1393,6 @@ RegBankLegalizeRules::RegBankLegalizeRules(const GCNSubtarget &_ST,
             {{VgprV2S16},
              {VgprV2S16, SgprV4S32_WF, Vgpr32, Vgpr32, Sgpr32_WF}}});
 
-  // The data is a VGPR value; the dword index is waterfalled into an SGPR.
   addRulesForGOpcs({G_AMDGPU_REG_LOAD}).Any({{BRC}, {{VgprBRC}, {Sgpr32_WF}}});
 
   addRulesForGOpcs({G_AMDGPU_REG_STORE})
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
index f2bd7eee43c74..e7a22d9e0c956 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
@@ -3090,8 +3090,6 @@ void AMDGPURegisterBankInfo::applyMappingImpl(
   }
   case AMDGPU::G_AMDGPU_REG_LOAD:
   case AMDGPU::G_AMDGPU_REG_STORE: {
-    // The dword index (operand 1) must be uniform; a divergent index needs a
-    // waterfall loop.
     applyDefaultMapping(OpdMapper);
     executeInWaterfallLoop(B, MI, {1});
     return;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSearchableTables.td b/llvm/lib/Target/AMDGPU/AMDGPUSearchableTables.td
index 9df51e02f826d..f295f794d8df2 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSearchableTables.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSearchableTables.td
@@ -430,10 +430,6 @@ def AMDGPUImageDMaskIntrinsicTable : GenericTable {
   let PrimaryKeyEarlyOut = 1;
 }
 
-//===----------------------------------------------------------------------===//
-// V_LOAD/STORE_IDX opcode mapping table.
-//===----------------------------------------------------------------------===//
-
 def VLdStIdxOpcodeInfoTable : GenericTable {
   let FilterClass = "VLdStIdxOpcodeInfo";
   let CppTypeName = "VLdStIdxOpcodeInfo";
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 4762a9f8b2d1d..6641fafa36110 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1208,8 +1208,8 @@ bool GCNTTIImpl::isSourceOfDivergence(const Value *V) const {
 
   // Loads from the private and flat address spaces are divergent, because
   // threads can execute the load instruction with the same inputs and get
-  // different results. The VGPR ("as memory") space is likewise divergent: it
-  // is a per-lane view, so even a uniform offset yields a per-lane value.
+  // different results. So are loads from the VGPR address space, since each
+  // lane reads its own registers.
   //
   // All other loads are not divergent, because if threads issue loads with the
   // same arguments, they will always get the same result.
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 78daff21baefa..f4e106d5e9342 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -9445,9 +9445,6 @@ SDValue SITargetLowering::LowerINLINEASM(SDValue Op, SelectionDAG &DAG) const {
 
 SDValue SITargetLowering::getSegmentAperture(unsigned AS, const SDLoc &DL,
                                              SelectionDAG &DAG) const {
-  // An address space with no aperture of its own round-trips through generic
-  // using the shared aperture tagged with its aperture number. Dereferencing
-  // such a pointer is UB; the round-trip only has to preserve the value.
   unsigned BaseAS = AS;
   unsigned SANum = AMDGPU::getSyntheticApertureNumber(AS);
   if (SANum != AMDGPU::SyntheticAperture::None)
@@ -13604,9 +13601,7 @@ static bool addressMayBeAccessedAsPrivate(const MachineMemOperand *MMO,
 }
 
 // Lower a VGPR ("as memory") load or store to a REG_LOAD / REG_STORE node
-// carrying the dword index (pointer >> 2).
-//
-// TODO: sub-dword (8/16-bit) accesses are diagnosed as unsupported below.
+// carrying the dword index. TODO: sub-dword accesses.
 SDValue SITargetLowering::LowerLoadStoreVGPR(SDValue Op,
                                              SelectionDAG &DAG) const {
   SDLoc DL(Op);
@@ -13614,8 +13609,7 @@ SDValue SITargetLowering::LowerLoadStoreVGPR(SDValue Op,
   EVT MemVT = MemOp->getMemoryVT();
   unsigned BitWidth = MemVT.getSizeInBits();
 
-  // Both callers replace the node with this result, so the diagnostic is
-  // emitted exactly once.
+  // Each caller replaces the node, so this is diagnosed once.
   auto reportUnsupported = [&]() -> SDValue {
     const Function &F = DAG.getMachineFunction().getFunction();
     DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
@@ -13634,9 +13628,7 @@ SDValue SITargetLowering::LowerLoadStoreVGPR(SDValue Op,
   // reach the containing dword rather than the bytes asked for.
   if (MemOp->getAlign() < Align(4))
     return reportUnsupported();
-  if (auto *Load = dyn_cast<LoadSDNode>(MemOp)) {
-    if (Load->getExtensionType() != ISD::NON_EXTLOAD)
-      return reportUnsupported();
+  if (isa<LoadSDNode>(MemOp)) {
     if (AMDGPUMI::VLoadIdxInst::tryGetOpcodeForBitWidth(BitWidth) == -1)
       return reportUnsupported();
   } else {
@@ -13673,9 +13665,17 @@ SDValue SITargetLowering::LowerLoadStoreVGPR(SDValue Op,
   SDValue NewLoad = DAG.getMemIntrinsicNode(
       AMDGPUISD::REG_LOAD, DL, DAG.getVTList(RegVT, MVT::Other), {Chain, Index},
       MemVT, LoadOp->getMemOperand());
-  if (RegVT == MemVT)
+  SDValue Value = NewLoad;
+  if (RegVT != MemVT)
+    Value = DAG.getNode(ISD::BITCAST, DL, MemVT, Value);
+  // The combiner folds an extension of a whole-dword load into the load.
+  EVT VT = LoadOp->getValueType(0);
+  if (VT != MemVT)
+    Value = DAG.getNode(ISD::getExtForLoadExtType(VT.isFloatingPoint(),
+                                                  LoadOp->getExtensionType()),
+                        DL, VT, Value);
+  if (Value == NewLoad)
     return NewLoad;
-  SDValue Value = DAG.getNode(ISD::BITCAST, DL, MemVT, NewLoad);
   return DAG.getMergeValues({Value, NewLoad.getValue(1)}, DL);
 }
 
@@ -20840,13 +20840,12 @@ bool SITargetLowering::isSDNodeSourceOfDivergence(const SDNode *N,
   case ISD::LOAD: {
     const LoadSDNode *L = cast<LoadSDNode>(N);
     unsigned AS = L->getAddressSpace();
-    // A VGPR "as memory" load reads this lane's own registers, so it is
-    // divergent however uniform the index is.
+    // A flat load may access private memory, and a VGPR load reads each lane's
+    // own registers.
     return AS == AMDGPUAS::PRIVATE_ADDRESS || AS == AMDGPUAS::FLAT_ADDRESS ||
            AS == AMDGPUAS::VGPR;
   }
-  // As above, after the pre-ISel combine. Without this a uniform index would
-  // make the loaded value look uniform and consumers would v_readfirstlane it.
+  // As above, after the pre-ISel combine.
   case AMDGPUISD::REG_LOAD:
     return true;
   case ISD::CALLSEQ_END:
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index 497db750b3bf2..25622a2c3f2a1 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -670,9 +670,8 @@ bool SIInstrInfo::getMemOperandsWithOffsetWidth(
   }
 
   if (auto *LdStIdx = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&LdSt)) {
-    // Callers treat identical base operands as the same address, which only
-    // holds while the base names a value. The movrel form reads M0, and M0 can
-    // be redefined between accesses, so report it as opaque.
+    // The movrel form's index is whatever M0 holds, which can change between
+    // accesses, so it has no base operand to report.
     if (!LdStIdx->isGPRIdx())
       return false;
     BaseOp = &LdStIdx->getIdxOp();
@@ -7791,16 +7790,13 @@ SIInstrInfo::legalizeOperands(MachineInstr &MI,
     return CreatedBB;
   }
 
-  // The VGPR indexing mode form takes its dword index in an SGPR, so a
-  // divergent index is made uniform with a waterfall loop. The movrel form
-  // reads M0, whose divergent writes are waterfalled when the copy into M0 is
-  // lowered.
+  // The GPR_IDX form needs its index in an SGPR. For the movrel form, a
+  // divergent M0 write is waterfalled where the copy into M0 is lowered.
   if (auto *LdStIdx = dyn_cast<AMDGPUMI::VLoadStoreIdxInst>(&MI)) {
     if (!LdStIdx->isGPRIdx())
       return CreatedBB;
     MachineOperand *Idx = &LdStIdx->getIdxOp();
-    // isSGPRReg handles physical registers too, so an unexpected physical index
-    // is waterfalled rather than silently skipped.
+    // A physical VGPR index is waterfalled too; isSGPRReg covers both.
     if (Idx->isReg() && !RI.isSGPRReg(MRI, Idx->getReg()))
       CreatedBB = generateWaterFallLoop(*this, MI, {Idx}, MDT);
     return CreatedBB;
@@ -8296,9 +8292,9 @@ void SIInstrInfo::handleCopyToPhysHelper(
   MachineBasicBlock::iterator I = Inst.getIterator();
   MachineBasicBlock::iterator E = Inst.getParent()->end();
   if (DstReg == AMDGPU::M0) {
-    // A VGPR "as memory" access indexes with M0 in every lane, so like the
-    // SGPR arguments of SI_CALL_ISEL below it is waterfalled. Other readers of
-    // M0 take the first lane.
+    // A VGPR "as memory" access uses M0 in every lane, so, like the SGPR
+    // arguments of SI_CALL_ISEL below, it is waterfalled. Other readers of M0
+    // take lane 0.
     SmallVector<MachineOperand *, 4> IdxOps;
     bool HasOtherReaders = false;
     while (++I != E) {
@@ -11355,15 +11351,14 @@ SIInstrInfo::getGenericValueUniformity(const MachineInstr &MI) const {
     return ValueUniformity::Default;
   }
 
-  // Always divergent: it reads the wave's per-lane registers, so even a uniform
-  // index yields a per-lane value.
+  // Each lane reads its own registers, however uniform the index.
   if (Opcode == AMDGPU::G_AMDGPU_REG_LOAD)
     return ValueUniformity::NeverUniform;
 
   // Loads from the private and flat address spaces are divergent, because
   // threads can execute the load instruction with the same inputs and get
-  // different results. The VGPR address space is likewise divergent (see
-  // above; this covers a G_LOAD not yet legalized to G_AMDGPU_REG_LOAD).
+  // different results. So are VGPR address space loads, before they become
+  // G_AMDGPU_REG_LOAD.
   //
   // All other loads are not divergent, because if threads issue loads with the
   // same arguments, they will always get the same result.
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index 26ee701495bfe..45a24f7db8977 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1064,9 +1064,7 @@ def SI_INDIRECT_DST_V32 : SI_INDIRECT_DST<VReg_1024>;
 // VGPR "as memory" indexed load/store pseudos (address space 13)
 //===----------------------------------------------------------------------===//
 
-// Populates the VLdStIdxOpcodeInfo searchable table, mapping each pseudo to its
-// bit width, load/store direction and indexing mechanism (see
-// AMDGPUMachineInstrs.h).
+// An entry in the VLdStIdxOpcodeInfo searchable table.
 class VLdStIdxOpcodeInfo<int size, bit isStore, bit isGPRIdx> {
   Instruction Opcode = !cast<Instruction>(NAME);
   bits<12> BitWidth = size;
@@ -1078,13 +1076,9 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
               VReg_224, VReg_256, VReg_288, VReg_320, VReg_352, VReg_384,
               VReg_512, VReg_1024] in {
   // $offset and the index are in dwords. As with V_INDIRECT_REG_*_MOVREL and
-  // V_INDIRECT_REG_*_GPR_IDX, the movrel form reads the index from M0, which
-  // the selection patterns write, and the VGPR indexing mode form takes it in
-  // an SGPR and clobbers M0 through the s_set_gpr_idx_on it expands to.
-  //
-  // Unlike those pseudos, the storage indexed here is the whole register file
-  // rather than a tuple an operand can name, so these are memory accesses
-  // carrying an MMO.
+  // _GPR_IDX, the movrel form reads the index from M0 and the GPR_IDX form
+  // takes it in an SGPR and clobbers M0. The storage is the whole register
+  // file, which no operand can name, so these are memory accesses with an MMO.
   def V_LOAD_IDX_B#rc.Size : VPseudoInstSI <
     (outs rc:$data),
     (ins i32imm:$offset)>,
@@ -1125,14 +1119,10 @@ foreach rc = [VGPR_32, VReg_64, VReg_96, VReg_128, VReg_160, VReg_192,
   }
 }
 
-// Copies of v_movrel[sd]_b32 for the moves the pseudos above expand into.
-//
-// The verifier requires a movrel to implicitly use the tuple it indexes. Here
-// the storage is the whole register file, which no operand can name, so these
-// carry their own opcodes; AMDGPUMCInstLower maps them back for encoding.
-// AMDGPULowerVGPREncoding reaches $vdst and $src0 by name, so these must be in
-// the named operand table. Without it no S_SET_VGPR_MSB is emitted and a base
-// register at or above 256 is silently encoded as its low eight bits.
+// v_movrel[sd]_b32 for the moves the pseudos above expand into. The verifier
+// requires a movrel to use the tuple it indexes, and no operand can name the
+// register file, so these have their own opcodes, mapped back by
+// AMDGPUMCInstLower. AMDGPULowerVGPREncoding needs their named operand table.
 let VOP1 = 1, Uses = [M0, EXEC], hasSideEffects = 0,
     UseNamedOperandTable = 1, Size = V_MOV_B32_e32.Size in {
   def V_MOVRELS_B32_as_mem
@@ -1141,9 +1131,8 @@ let VOP1 = 1, Uses = [M0, EXEC], hasSideEffects = 0,
       : VPseudoInstSI<(outs), (ins VGPR_32:$vdst, VSrc_b32:$src0)>;
 }
 
-// An (add idx, imm) shape folds a constant dword offset into $offset. In the
-// movrel form the index is an M0 input, so the copy into M0 is emitted by
-// selection itself; a divergent one is waterfalled by SIFixSGPRCopies.
+// An (add idx, imm) index folds the constant into $offset. The movrel patterns
+// take the index as an M0 input, so selection itself writes M0.
 multiclass VRegIdxLoadStorePat<ValueType vt> {
   defvar load_inst = !cast<Instruction>("V_LOAD_IDX_B"#vt.Size);
   defvar store_inst = !cast<Instruction>("V_STORE_IDX_B"#vt.Size);
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll
index 17b80a07f283a..fb0475f4b2ab3 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-addrspacecast.ll
@@ -4,10 +4,9 @@
 ; RUN: llc -global-isel=0 -mtriple=amdgpu12.50-- < %s | FileCheck %s --check-prefixes=GFX1250,GFX1250-SDAG
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.50-- < %s | FileCheck %s --check-prefixes=GFX1250,GFX1250-GISEL
 
-; A pointer to the VGPR "as memory" address space (13) round-trips through the
-; generic address space using a synthetic aperture: the shared aperture with the
-; aperture number in its low bits. The round-trip preserves the value, including
-; the -1 null pointer, but the generic pointer must not be dereferenced.
+; VGPR "as memory" (address space 13) pointers round-trip through the generic
+; address space via a synthetic aperture, preserving the value, including the
+; -1 null pointer.
 
 define ptr @vgpr_to_flat(ptr addrspace(13) %ptr) {
 ; GFX12-SDAG-LABEL: vgpr_to_flat:
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-atomic.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-atomic.ll
index 9f9d206e4879a..341db4dcd2fa2 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-atomic.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-atomic.ll
@@ -2,9 +2,8 @@
 ; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
 
-; Only the executing lane can access its view of the VGPRs, so, as for the
-; private address space, atomics on the VGPR "as memory" address space (13) are
-; performed as ordinary loads and stores.
+; Only the executing lane can reach its VGPRs, so, as for private memory,
+; atomics on address space 13 are ordinary loads and stores.
 
 define i32 @atomicrmw_add(ptr addrspace(13) inreg %p, i32 %v) {
 ; GFX12-LABEL: atomicrmw_add:
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
index c14730105de3b..ee7d5ca2ec4f4 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-basic.ll
@@ -2,10 +2,8 @@
 ; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
 
-; End-to-end lowering of the VGPR "as memory" address space (13) on a
-; movrel-capable subtarget (gfx12). A load/store of a uniform (SGPR) pointer
-; lowers to an M0-relative move (v_movrels_b32 / v_movreld_b32) over the wave's
-; vector registers, with the dword index (pointer >> 2) placed in M0.
+; End-to-end lowering on gfx12: an access through a uniform pointer is an
+; M0-relative v_movrels_b32 / v_movreld_b32, with the dword index in M0.
 
 define i32 @load_i32(ptr addrspace(13) inreg %p) {
 ; GFX12-LABEL: load_i32:
@@ -320,9 +318,7 @@ define i32 @load_i32_twice(ptr addrspace(13) inreg %p) {
   ret i32 %z
 }
 
-; Null and poison pointers must be accepted (produce valid code) rather than
-; crash or fail the machine verifier. The specific null-pointer value is
-; defined by the parent change that introduces the address space.
+; Null and poison pointers must produce valid code.
 
 define i32 @load_null() {
 ; GFX12-LABEL: load_null:
@@ -405,3 +401,39 @@ define void @store_poison(i32 %v) {
   store i32 %v, ptr addrspace(13) poison
   ret void
 }
+
+; An extension of a whole-dword load is folded into the load by the combiner.
+define i64 @load_zext_i64(ptr addrspace(13) inreg %p) {
+; GFX12-LABEL: load_zext_i64:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_mov_b32_e32 v1, 0
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %v = load i32, ptr addrspace(13) %p
+  %z = zext i32 %v to i64
+  ret i64 %z
+}
+
+define i64 @load_sext_i64(ptr addrspace(13) inreg %p) {
+; GFX12-LABEL: load_sext_i64:
+; GFX12:       ; %bb.0:
+; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT:    s_wait_expcnt 0x0
+; GFX12-NEXT:    s_wait_samplecnt 0x0
+; GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT:    v_ashrrev_i32_e32 v1, 31, v0
+; GFX12-NEXT:    s_setpc_b64 s[30:31]
+  %v = load i32, ptr addrspace(13) %p
+  %s = sext i32 %v to i64
+  ret i64 %s
+}
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir
index 73d691c0c7473..754f61f196178 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir
@@ -1,9 +1,8 @@
 # NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6
 # RUN: llc -mtriple=amdgpu12.00-- -run-pass=si-fix-sgpr-copies -o - %s | FileCheck %s
 
-# A VGPR "as memory" access on a movrel subtarget reads its index from M0 in
-# every lane, so a divergent value copied into M0 for it is waterfalled rather
-# than taken from the first lane.
+# A movrel access reads M0 in every lane, so a divergent value copied into M0
+# for it is waterfalled rather than taken from the first lane.
 
 ---
 name:            divergent_load
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
index a69908541ff71..b694b6542d688 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent.ll
@@ -4,16 +4,10 @@
 ; RUN: llc -global-isel=0 -mtriple=amdgpu9.42-- < %s | FileCheck %s --check-prefixes=GFX942,GFX942-SDAG
 ; RUN: llc -global-isel=1 -mtriple=amdgpu9.42-- < %s | FileCheck %s --check-prefixes=GFX942,GFX942-GISEL
 
-; A divergent (per-lane) dword index into VGPR "as memory" (address space 13)
-; is handled with a waterfall loop: for each unique index across the wave, set
-; M0 and do the M0-relative move under a matching-lane EXEC subset. The pointer
-; arrives in a VGPR (no inreg), so the index (pointer >> 2) is divergent.
-;
-; gfx942 covers the two axes gfx1200 cannot. It is wave64, so the loop runs over
-; a 64-lane mask rather than a 32-lane one, and it indexes with the VGPR indexing
-; mode instead of movrel, so the index reaches the hardware by a different route.
-; gfx1250 cannot stand in for the first: it is wave32 only, and asking it for
-; wave64 makes llc emit no functions at all, which reads as a passing test.
+; A divergent index (the pointer is in a VGPR) is waterfalled: one iteration per
+; unique index, with the move done for the matching lanes. gfx942 adds wave64
+; and the VGPR indexing mode; gfx1250 cannot cover wave64, since it is wave32
+; only and llc emits no functions when asked for wave64.
 
 define i32 @load_i32(ptr addrspace(13) %p) {
 ; GFX12-SDAG-LABEL: load_i32:
@@ -220,10 +214,8 @@ define void @store_i32(ptr addrspace(13) %p, i32 %x) {
   store i32 %y, ptr addrspace(13) %p
   ret void
 }
-; The value read out of the address space is per-lane whatever the index is, so
-; a uniform index does not make it uniform. A consumer that requires a uniform
-; operand must therefore not be handed it directly: doing so inserts a
-; readfirstlane, which broadcasts one lane's value across the whole wave.
+; The loaded value is per-lane even with a uniform index, so a consumer that
+; needs a uniform operand must not get it by readfirstlane.
 declare i32 @llvm.amdgcn.s.buffer.load.i32(<4 x i32>, i32, i32 immarg)
 
 define i32 @uniform_index_divergent_value(ptr addrspace(13) inreg %p, <4 x i32> inreg %rsrc) {
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
index 25b865f5478ca..5ba57644d4655 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-gpr-idx-mode.ll
@@ -4,14 +4,8 @@
 ; RUN: llc -global-isel=0 -mtriple=amdgpu9.0a-- -filetype=null < %s
 ; RUN: llc -global-isel=1 -mtriple=amdgpu9.0a-- -filetype=null < %s
 
-; The VGPR "as memory" address space (13) on a subtarget that has no movrel.
-; gfx9 indexes with the VGPR indexing mode instead, so AMDGPULowerVGPREncoding
-; wraps the move in s_set_gpr_idx_on / s_set_gpr_idx_off, which takes the dword
-; index straight from the SGPR holding it. Nothing needs to copy the index into
-; M0, so AMDGPUAssignIdxToM0 does nothing on these subtargets.
-;
-; The mode switch and the moves it applies to are bundled, so nothing can be
-; scheduled or spilled between them while indexing is enabled.
+; gfx9 has no movrel, so the moves are wrapped in s_set_gpr_idx_on / _off, which
+; take the dword index from its SGPR, and bundled with them.
 
 define i32 @load_i32(ptr addrspace(13) inreg %p) {
 ; GFX9-LABEL: load_i32:
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-high-registers.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-high-registers.ll
index be5c4d53cd31d..083c01f9a9ee4 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-high-registers.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-high-registers.ll
@@ -1,20 +1,10 @@
 ; RUN: llc -global-isel=0 -mtriple=amdgpu12.50-- < %s | FileCheck %s --check-prefixes=CHECK,SDAG
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.50-- < %s | FileCheck %s --check-prefixes=CHECK,GISEL
 
-; A subtarget with more than 256 addressable VGPRs encodes a register number's
-; high bits separately, with S_SET_VGPR_MSB. A whole-dword access folds its
-; constant dword offset into the base register of each indexed move, so an
-; access whose base reaches v256 or beyond needs those bits described -
-; otherwise only the low eight bits are encoded and the move silently touches a
-; register 256 lower than the one meant.
-;
-; The moves must therefore be in the named operand table, since that is how
-; AMDGPULowerVGPREncoding finds the operands whose high bits it has to describe.
-;
-; Only SelectionDAG folds the offset into the base; GlobalISel folds it into the
-; index instead and indexes from v0, so it cannot reach a high base this way and
-; needs no mode change. Both are checked, because the difference is the reason
-; this went unnoticed.
+; With more than 256 VGPRs, a move whose base register is v256 or above needs
+; S_SET_VGPR_MSB, or only the low eight bits are encoded. SelectionDAG can fold
+; a constant offset into the base register and so reach a high base; GlobalISel
+; folds it into the index and always uses v0 as the base.
 
 ; The dword index is %i + 254, so a four-dword access spans v254, v255, v256 and
 ; v257 relative to M0.
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll
index db1d8ad0bd4f6..d4a9ce6f07d01 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-index-demanded-bits.ll
@@ -2,11 +2,9 @@
 ; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
 
-; The VGPR "as memory" (address space 13) dword index only needs enough bits to
-; address all addressable VGPRs, so a redundant high-bit mask feeding the index
-; folds away via the AMDGPUISD::REG_LOAD / REG_STORE SimplifyDemandedBits combine.
-; The incoming index is masked with 0xffff (wider than necessary); on the SDAG
-; path the mask must not survive into the M0 index computation.
+; The dword index only needs enough bits to address every VGPR, so a wider mask
+; on it folds away (SimplifyDemandedBits on REG_LOAD / REG_STORE): on
+; SelectionDAG the 0xffff mask must not reach the M0 computation.
 
 define amdgpu_ps i32 @load_masked_index(i32 inreg %arg) {
 ; GFX12-SDAG-LABEL: load_masked_index:
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll
index cf9616322fd73..a4842bd2b3e34 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-optnone.ll
@@ -2,11 +2,8 @@
 ; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12
 
-; AMDGPUAssignIdxToM0 is required lowering rather than an optimization: the
-; v_movrel that AMDGPULowerVGPREncoding emits reads the dword index from M0, so
-; without the copy to M0 it reads a stale value. The pass must therefore run
-; even for optnone functions - clang marks every function optnone at -O0 - so
-; the index below has to end up in M0 and not in a plain SGPR.
+; clang marks every function optnone at -O0; accesses there must still be
+; lowered, with the index in M0.
 
 define i32 @load_i32_optnone(ptr addrspace(13) inreg %p) noinline optnone {
 ; GFX12-LABEL: load_i32_optnone:
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
index 52fd329aa1428..739ad886bd2ca 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-unsupported.ll
@@ -1,13 +1,10 @@
 ; RUN: not llc -global-isel=0 -mtriple=amdgpu12.00-- -filetype=null < %s 2>&1 | FileCheck %s
 ; RUN: not llc -global-isel=1 -mtriple=amdgpu12.00-- -filetype=null < %s 2>&1 | FileCheck %s
 
-; Accesses of the VGPR "as memory" address space (13) that are not implemented
-; must be rejected with a clean diagnostic on both SelectionDAG and GlobalISel,
-; rather than failing with "cannot select" / "unable to legalize" - or, worse,
-; silently generating wrong code.
+; Unimplemented accesses get a clean diagnostic on both selectors, not a
+; selection or legalization failure, or wrong code.
 
-; Sub-dword (8/16-bit) accesses are not yet implemented; support lands in a
-; later change.
+; Sub-dword accesses are not implemented yet.
 ; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only dword-aligned whole-dword loads and stores are implemented
 define i8 @load_i8(ptr addrspace(13) inreg %p) {
   %x = load i8, ptr addrspace(13) %p
@@ -32,11 +29,8 @@ define void @store_i16(ptr addrspace(13) inreg %p, i16 %v) {
   ret void
 }
 
-; An access addresses registers by the dword index pointer >> 2, which discards
-; the low two bits rather than accounting for them. An under-aligned one would
-; therefore reach the dword containing the address instead of the bytes asked
-; for - the same code as a correctly aligned access, reading the wrong data with
-; nothing to show for it.
+; The index is the pointer >> 2, so an under-aligned access would silently reach
+; the containing dword.
 ; CHECK: error: {{.*}}unsupported access of VGPR 'as memory' address space (13); only dword-aligned whole-dword loads and stores are implemented
 define i32 @load_i32_align1(ptr addrspace(13) inreg %p) {
   %x = load i32, ptr addrspace(13) %p, align 1
@@ -49,8 +43,7 @@ define void @store_i32_align1(ptr addrspace(13) inreg %p, i32 %v) {
   ret void
 }
 
-; Alignment is required of the pointer, not of the accessed type: a 64-bit
-; access needs only the dword alignment the index computation relies on.
+; Only dword alignment is required, whatever the access size.
 ; CHECK-NOT: in function load_i64_align4
 define i64 @load_i64_align4(ptr addrspace(13) inreg %p) {
   %x = load i64, ptr addrspace(13) %p, align 4
diff --git a/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp b/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp
index 8f5b191754d40..e4494dcc253bd 100644
--- a/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp
+++ b/llvm/unittests/Target/AMDGPU/VGPRAsMemory.cpp
@@ -6,12 +6,9 @@
 //
 //===----------------------------------------------------------------------===//
 //
-// Two properties of the VGPR "as memory" (address space 13) indexed accesses
-// are asserted here rather than in a lit test, because no pass can be made to
-// observe them: these pseudos read M0, which is written in any function that
-// uses them, and that alone stops MachineLICM and MachineSink from moving
-// them. Their safety today is therefore incidental, and the properties below
-// are what it would rest on if that incidental protection ever went away.
+// Properties of the VGPR "as memory" (address space 13) accesses that no lit
+// test can observe: the M0 they read already stops MachineLICM and MachineSink
+// from moving them. These checks are what remains if that ever changes.
 //
 //===----------------------------------------------------------------------===//
 
@@ -31,10 +28,8 @@ class VGPRAsMemoryTest : public AMDGPUCodeGenTestBase {
   void SetUp() override { setUpImpl("amdgpu12.00-amd-", "", ""); }
 };
 
-// An indexed access reads or writes the per-lane vector registers of the active
-// lanes, so which lanes are active is part of what it does. Its implicit use of
-// EXEC must not be reported ignorable: that is what would otherwise let it be
-// hoisted or sunk across a write to EXEC, changing the set of lanes touched.
+// An access touches only the active lanes' registers, so its EXEC use must not
+// be ignorable, or it could be moved across a write to EXEC.
 TEST_F(VGPRAsMemoryTest, ExecUseIsNotIgnorable) {
   StringRef MIRString = R"MIR(
 name: exec_use
@@ -86,12 +81,8 @@ body:             |
   }
 }
 
-// Two whole-dword accesses indexed by M0 - the form every access has on a
-// movrel subtarget - with M0 redefined between them. The dwords they touch are
-// (first M0)+1 and (second M0)+0, which are the same dword whenever the second
-// index is one more than the first, so they may alias. Disjointness is decided
-// from a base operand and the constant offset, and these have no base operand
-// that could tell the two M0 values apart.
+// With M0 redefined between them, offsets 1 and 0 can still name the same
+// dword, and nothing but M0 tells the two indices apart, so these may alias.
 TEST_F(VGPRAsMemoryTest, M0IndexedAccessesAcrossAM0RedefMayAlias) {
   StringRef MIRString = R"MIR(
 name: m0_redef

>From 985a821879ea148b9aeb8393c61f93c2e39f2e8c Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Fri, 25 Sep 2026 13:36:34 -0400
Subject: [PATCH 31/35] Regenerate as-vgpr-divergent-m0.mir now that the
 waterfall marks SCC dead

---
 .../AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir | 12 ++++++------
 1 file changed, 6 insertions(+), 6 deletions(-)

diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir
index 754f61f196178..f05cb6586bbb7 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-divergent-m0.mir
@@ -31,7 +31,7 @@ body:             |
     ; CHECK-NEXT: {{  $}}
     ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_]]
     ; CHECK-NEXT: [[V_LOAD_IDX_B32_:%[0-9]+]]:vgpr_32 = V_LOAD_IDX_B32 1, implicit killed $m0, implicit $exec :: (load (s32), addrspace 13)
-    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def $scc
+    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def dead $scc
     ; CHECK-NEXT: $exec_lo = S_MOV_B32_term [[S_ANDN2_B32_]]
     ; CHECK-NEXT: SI_WATERFALL_LOOP %bb.1, implicit $exec
     ; CHECK-NEXT: {{  $}}
@@ -74,7 +74,7 @@ body:             |
     ; CHECK-NEXT: {{  $}}
     ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_]]
     ; CHECK-NEXT: V_STORE_IDX_B32 [[COPY1]], 2, implicit killed $m0, implicit $exec :: (store (s32), addrspace 13)
-    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def $scc
+    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def dead $scc
     ; CHECK-NEXT: $exec_lo = S_MOV_B32_term [[S_ANDN2_B32_]]
     ; CHECK-NEXT: SI_WATERFALL_LOOP %bb.1, implicit $exec
     ; CHECK-NEXT: {{  $}}
@@ -117,7 +117,7 @@ body:             |
     ; CHECK-NEXT: {{  $}}
     ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_]]
     ; CHECK-NEXT: [[V_LOAD_IDX_B32_:%[0-9]+]]:vgpr_32 = V_LOAD_IDX_B32 0, implicit killed $m0, implicit $exec :: (load (s32), addrspace 13)
-    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def $scc
+    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def dead $scc
     ; CHECK-NEXT: $exec_lo = S_MOV_B32_term [[S_ANDN2_B32_]]
     ; CHECK-NEXT: SI_WATERFALL_LOOP %bb.1, implicit $exec
     ; CHECK-NEXT: {{  $}}
@@ -163,7 +163,7 @@ body:             |
     ; CHECK-NEXT: {{  $}}
     ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_1]]
     ; CHECK-NEXT: [[V_LOAD_IDX_B32_:%[0-9]+]]:vgpr_32 = V_LOAD_IDX_B32 0, implicit killed $m0, implicit $exec :: (load (s32), addrspace 13)
-    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def $scc
+    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def dead $scc
     ; CHECK-NEXT: $exec_lo = S_MOV_B32_term [[S_ANDN2_B32_]]
     ; CHECK-NEXT: SI_WATERFALL_LOOP %bb.1, implicit $exec
     ; CHECK-NEXT: {{  $}}
@@ -207,7 +207,7 @@ body:             |
     ; CHECK-NEXT: {{  $}}
     ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_]]
     ; CHECK-NEXT: [[V_LOAD_IDX_B32_:%[0-9]+]]:vgpr_32 = V_LOAD_IDX_B32 0, implicit killed $m0, implicit $exec :: (load (s32), addrspace 13)
-    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def $scc
+    ; CHECK-NEXT: [[S_ANDN2_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI]], $exec_lo, implicit-def dead $scc
     ; CHECK-NEXT: $exec_lo = S_MOV_B32_term [[S_ANDN2_B32_]]
     ; CHECK-NEXT: SI_WATERFALL_LOOP %bb.4, implicit $exec
     ; CHECK-NEXT: {{  $}}
@@ -230,7 +230,7 @@ body:             |
     ; CHECK-NEXT: {{  $}}
     ; CHECK-NEXT: $m0 = COPY [[V_READFIRSTLANE_B32_1]]
     ; CHECK-NEXT: [[V_LOAD_IDX_B32_1:%[0-9]+]]:vgpr_32 = V_LOAD_IDX_B32 0, implicit killed $m0, implicit $exec :: (load (s32), addrspace 13)
-    ; CHECK-NEXT: [[S_ANDN2_B32_1:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI1]], $exec_lo, implicit-def $scc
+    ; CHECK-NEXT: [[S_ANDN2_B32_1:%[0-9]+]]:sreg_32_xm0_xexec = S_ANDN2_B32 [[PHI1]], $exec_lo, implicit-def dead $scc
     ; CHECK-NEXT: $exec_lo = S_MOV_B32_term [[S_ANDN2_B32_1]]
     ; CHECK-NEXT: SI_WATERFALL_LOOP %bb.1, implicit $exec
     ; CHECK-NEXT: {{  $}}

>From 30ed8db8de520da14aae687a1be82b9fd5906ae3 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Tue, 29 Sep 2026 17:53:10 -0400
Subject: [PATCH 32/35] Drop a redundant comment in lowerLoadStoreVGPR

---
 llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp | 1 -
 1 file changed, 1 deletion(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index 301618170276f..59449d26236e0 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -3506,7 +3506,6 @@ static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
 
   const LLT ValTy = MRI.getType(ValReg);
   const unsigned ValSize = ValTy.getSizeInBits();
-  // The selection patterns match integer LLTs rather than plain scalars.
   const LLT I32 = LLT::integer(32);
 
   // The index is the pointer >> 2, so an under-aligned access would silently

>From 6e90a3d9738cbccfd9e77008027224c20e19f412 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Tue, 29 Sep 2026 18:10:23 -0400
Subject: [PATCH 33/35] Use GLoadStore

---
 .../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp | 22 +++++++++----------
 1 file changed, 11 insertions(+), 11 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index 59449d26236e0..796def757d695 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -3495,14 +3495,14 @@ static LLT widenToNextPowerOf2(LLT Ty) {
 
 /// Lower a whole-dword G_LOAD / G_STORE on AMDGPUAS::VGPR into
 /// G_AMDGPU_REG_LOAD / G_AMDGPU_REG_STORE. Parallels LowerLoadStoreVGPR.
-static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
+static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, GLoadStore &LdSt) {
   MachineIRBuilder &B = Helper.MIRBuilder;
   MachineRegisterInfo &MRI = *B.getMRI();
-  MachineMemOperand &MMO = **MI.memoperands_begin();
+  MachineMemOperand &MMO = LdSt.getMMO();
 
-  const bool IsStore = MI.getOpcode() == AMDGPU::G_STORE;
-  Register ValReg = MI.getOperand(0).getReg();
-  Register PtrReg = MI.getOperand(1).getReg();
+  const bool IsStore = isa<GStore>(LdSt);
+  Register ValReg = LdSt.getReg(0);
+  Register PtrReg = LdSt.getPointerReg();
 
   const LLT ValTy = MRI.getType(ValReg);
   const unsigned ValSize = ValTy.getSizeInBits();
@@ -3512,16 +3512,16 @@ static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
   // reach the containing dword.
   if (!isVGPRLoadStoreSizeSupported(MMO.getMemoryType().getSizeInBits(),
                                     ValSize) ||
-      MMO.getAlign() < Align(4)) {
+      LdSt.getAlign() < Align(4)) {
     const Function &F = B.getMF().getFunction();
     F.getContext().diagnose(DiagnosticInfoUnsupported(
         F,
         "unsupported access of VGPR 'as memory' address space (13); only "
         "dword-aligned whole-dword loads and stores are implemented",
-        MI.getDebugLoc()));
+        LdSt.getDebugLoc()));
     if (!IsStore)
       B.buildUndef(ValReg);
-    MI.eraseFromParent();
+    LdSt.eraseFromParent();
     return true;
   }
 
@@ -3552,7 +3552,7 @@ static bool lowerLoadStoreVGPR(LegalizerHelper &Helper, MachineInstr &MI) {
     B.buildBitcast(ValReg, Result);
   }
 
-  MI.eraseFromParent();
+  LdSt.eraseFromParent();
   return true;
 }
 
@@ -3567,7 +3567,7 @@ bool AMDGPULegalizerInfo::legalizeLoad(LegalizerHelper &Helper,
   unsigned AddrSpace = PtrTy.getAddressSpace();
 
   if (AddrSpace == AMDGPUAS::VGPR && MI.getOpcode() == AMDGPU::G_LOAD)
-    return lowerLoadStoreVGPR(Helper, MI);
+    return lowerLoadStoreVGPR(Helper, cast<GLoadStore>(MI));
 
   if (AddrSpace == AMDGPUAS::CONSTANT_ADDRESS_32BIT) {
     LLT ConstPtr = LLT::pointer(AMDGPUAS::CONSTANT_ADDRESS, 64);
@@ -3659,7 +3659,7 @@ bool AMDGPULegalizerInfo::legalizeStore(LegalizerHelper &Helper,
 
   if (MRI.getType(MI.getOperand(1).getReg()).getAddressSpace() ==
       AMDGPUAS::VGPR)
-    return lowerLoadStoreVGPR(Helper, MI);
+    return lowerLoadStoreVGPR(Helper, cast<GLoadStore>(MI));
 
   if (hasBufferRsrcWorkaround(DataTy)) {
     Observer.changingInstr(MI);

>From 3b0826f5dc995721d727ac54d8c12a180e32655b Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Wed, 30 Sep 2026 08:30:49 -0400
Subject: [PATCH 34/35] Don't assert on an out-of-range VGPR offset

---
 .../Target/AMDGPU/AMDGPULowerVGPREncoding.cpp |   5 -
 .../AddressSpaceVGPR/as-vgpr-out-of-range.ll  | 115 ++++++++++++++++++
 2 files changed, 115 insertions(+), 5 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-out-of-range.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
index cef26a266c60c..4d5b9b76e1a97 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
@@ -399,11 +399,6 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
   // addressable range below rather than emit an invalid register.
   unsigned NumAddressableVGPRs = ST->getAddressableNumVGPRs(
       MI.getMF()->getInfo<SIMachineFunctionInfo>()->getDynamicVGPRBlockSize());
-#ifndef NDEBUG
-  bool AllowOffsetWrap = NumAddressableVGPRs == ST->getTotalNumVGPRs();
-  assert((AllowOffsetWrap || Offset + NumDwords <= NumAddressableVGPRs) &&
-         "out of bounds VGPR 'as memory' (address space 13) access");
-#endif
 
   const bool UseGPRIdxMode = LdSt.isGPRIdx();
 
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-out-of-range.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-out-of-range.ll
new file mode 100644
index 0000000000000..b353b895b1798
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-out-of-range.ll
@@ -0,0 +1,115 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+
+; An access past the addressable VGPRs is undefined behavior but must still
+; compile. SelectionDAG folds these constant offsets into the move's base
+; register, which is masked into the addressable range.
+
+define void @load_past_end(ptr addrspace(1) %out, i32 inreg %i) {
+; GFX12-SDAG-LABEL: load_past_end:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_mov_b32 m0, s0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v44
+; GFX12-SDAG-NEXT:    global_store_b32 v[0:1], v2, off
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: load_past_end:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshl2_add_u32 s0, s0, 0x4b0
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v0
+; GFX12-GISEL-NEXT:    global_store_b32 v[0:1], v2, off
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %s = shl nuw i32 %i, 2
+  %a = add nuw i32 %s, 1200
+  %p = inttoptr i32 %a to ptr addrspace(13)
+  %v = load i32, ptr addrspace(13) %p, align 4
+  store i32 %v, ptr addrspace(1) %out, align 4
+  ret void
+}
+
+define void @store_past_end(i32 inreg %i, i32 %v) {
+; GFX12-SDAG-LABEL: store_past_end:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_mov_b32 m0, s0
+; GFX12-SDAG-NEXT:    v_movreld_b32_e32 v44, v0
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: store_past_end:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshl2_add_u32 s0, s0, 0x4b0
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-GISEL-NEXT:    v_movreld_b32_e32 v0, v0
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %s = shl nuw i32 %i, 2
+  %a = add nuw i32 %s, 1200
+  %p = inttoptr i32 %a to ptr addrspace(13)
+  store i32 %v, ptr addrspace(13) %p, align 4
+  ret void
+}
+
+define void @load_straddling_end(ptr addrspace(1) %out, i32 inreg %i) {
+; GFX12-SDAG-LABEL: load_straddling_end:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_mov_b32 m0, s0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v2, v254
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v3, v255
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v4, v0
+; GFX12-SDAG-NEXT:    v_movrels_b32_e32 v5, v1
+; GFX12-SDAG-NEXT:    global_store_b128 v[0:1], v[2:5], off
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: load_straddling_end:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_lshl2_add_u32 s0, s0, 0x3f8
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s0, 2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v0
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v3, v1
+; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v4, v2
+; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v5, v3
+; GFX12-GISEL-NEXT:    global_store_b128 v[0:1], v[2:5], off
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+  %s = shl nuw i32 %i, 2
+  %a = add nuw i32 %s, 1016
+  %p = inttoptr i32 %a to ptr addrspace(13)
+  %v = load <4 x i32>, ptr addrspace(13) %p, align 16
+  store <4 x i32> %v, ptr addrspace(1) %out, align 16
+  ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX12: {{.*}}

>From 04ad22f81a9213ff54d684ad86d098cd04d505b7 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Wed, 30 Sep 2026 08:45:39 -0400
Subject: [PATCH 35/35] Wrap an out-of-range VGPR offset at the architectural
 VGPR count

---
 .../Target/AMDGPU/AMDGPULowerVGPREncoding.cpp | 10 ++-
 .../AddressSpaceVGPR/as-vgpr-out-of-range.ll  | 71 +++++++++++++++++++
 2 files changed, 78 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
index 4d5b9b76e1a97..36d6c65ff7b1f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerVGPREncoding.cpp
@@ -396,9 +396,13 @@ void AMDGPULowerVGPREncoding::lowerLoadStoreIdx(MachineInstr &MI) {
   unsigned NumDwords = LdSt.getBitWidth() / 32;
 
   // A statically out-of-range offset is undefined behavior; mask it into the
-  // addressable range below rather than emit an invalid register.
-  unsigned NumAddressableVGPRs = ST->getAddressableNumVGPRs(
-      MI.getMF()->getInfo<SIMachineFunctionInfo>()->getDynamicVGPRBlockSize());
+  // addressable range below rather than emit an invalid register. The AGPRs
+  // that gfx90a+ counts as addressable cannot be named as VGPRs.
+  unsigned DynamicVGPRBlockSize =
+      MI.getMF()->getInfo<SIMachineFunctionInfo>()->getDynamicVGPRBlockSize();
+  unsigned NumAddressableVGPRs =
+      std::min(ST->getAddressableNumArchVGPRs(),
+               ST->getAddressableNumVGPRs(DynamicVGPRBlockSize));
 
   const bool UseGPRIdxMode = LdSt.isGPRIdx();
 
diff --git a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-out-of-range.ll b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-out-of-range.ll
index b353b895b1798..b2a854b512dca 100644
--- a/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-out-of-range.ll
+++ b/llvm/test/CodeGen/AMDGPU/AddressSpaceVGPR/as-vgpr-out-of-range.ll
@@ -1,6 +1,8 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
 ; RUN: llc -global-isel=0 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-SDAG
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.00-- < %s | FileCheck %s --check-prefixes=GFX12,GFX12-GISEL
+; RUN: llc -global-isel=0 -mtriple=amdgpu9.42-- < %s | FileCheck %s --check-prefixes=GFX942,GFX942-SDAG
+; RUN: llc -global-isel=1 -mtriple=amdgpu9.42-- < %s | FileCheck %s --check-prefixes=GFX942,GFX942-GISEL
 
 ; An access past the addressable VGPRs is undefined behavior but must still
 ; compile. SelectionDAG folds these constant offsets into the move's base
@@ -32,6 +34,28 @@ define void @load_past_end(ptr addrspace(1) %out, i32 inreg %i) {
 ; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v2, v0
 ; GFX12-GISEL-NEXT:    global_store_b32 v[0:1], v2, off
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX942-SDAG-LABEL: load_past_end:
+; GFX942-SDAG:       ; %bb.0:
+; GFX942-SDAG-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX942-SDAG-NEXT:    s_set_gpr_idx_on s0, gpr_idx(SRC0)
+; GFX942-SDAG-NEXT:    v_mov_b32_e32 v2, v44
+; GFX942-SDAG-NEXT:    s_set_gpr_idx_off
+; GFX942-SDAG-NEXT:    global_store_dword v[0:1], v2, off
+; GFX942-SDAG-NEXT:    s_waitcnt vmcnt(0)
+; GFX942-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX942-GISEL-LABEL: load_past_end:
+; GFX942-GISEL:       ; %bb.0:
+; GFX942-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX942-GISEL-NEXT:    s_lshl2_add_u32 s0, s0, 0x4b0
+; GFX942-GISEL-NEXT:    s_lshr_b32 s0, s0, 2
+; GFX942-GISEL-NEXT:    s_set_gpr_idx_on s0, gpr_idx(SRC0)
+; GFX942-GISEL-NEXT:    v_mov_b32_e32 v2, v0
+; GFX942-GISEL-NEXT:    s_set_gpr_idx_off
+; GFX942-GISEL-NEXT:    global_store_dword v[0:1], v2, off
+; GFX942-GISEL-NEXT:    s_waitcnt vmcnt(0)
+; GFX942-GISEL-NEXT:    s_setpc_b64 s[30:31]
   %s = shl nuw i32 %i, 2
   %a = add nuw i32 %s, 1200
   %p = inttoptr i32 %a to ptr addrspace(13)
@@ -64,6 +88,24 @@ define void @store_past_end(i32 inreg %i, i32 %v) {
 ; GFX12-GISEL-NEXT:    s_lshr_b32 m0, s0, 2
 ; GFX12-GISEL-NEXT:    v_movreld_b32_e32 v0, v0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX942-SDAG-LABEL: store_past_end:
+; GFX942-SDAG:       ; %bb.0:
+; GFX942-SDAG-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX942-SDAG-NEXT:    s_set_gpr_idx_on s0, gpr_idx(DST)
+; GFX942-SDAG-NEXT:    v_mov_b32_e32 v44, v0
+; GFX942-SDAG-NEXT:    s_set_gpr_idx_off
+; GFX942-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX942-GISEL-LABEL: store_past_end:
+; GFX942-GISEL:       ; %bb.0:
+; GFX942-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX942-GISEL-NEXT:    s_lshl2_add_u32 s0, s0, 0x4b0
+; GFX942-GISEL-NEXT:    s_lshr_b32 s0, s0, 2
+; GFX942-GISEL-NEXT:    s_set_gpr_idx_on s0, gpr_idx(DST)
+; GFX942-GISEL-NEXT:    v_mov_b32_e32 v0, v0
+; GFX942-GISEL-NEXT:    s_set_gpr_idx_off
+; GFX942-GISEL-NEXT:    s_setpc_b64 s[30:31]
   %s = shl nuw i32 %i, 2
   %a = add nuw i32 %s, 1200
   %p = inttoptr i32 %a to ptr addrspace(13)
@@ -104,6 +146,34 @@ define void @load_straddling_end(ptr addrspace(1) %out, i32 inreg %i) {
 ; GFX12-GISEL-NEXT:    v_movrels_b32_e32 v5, v3
 ; GFX12-GISEL-NEXT:    global_store_b128 v[0:1], v[2:5], off
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX942-SDAG-LABEL: load_straddling_end:
+; GFX942-SDAG:       ; %bb.0:
+; GFX942-SDAG-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX942-SDAG-NEXT:    s_set_gpr_idx_on s0, gpr_idx(SRC0)
+; GFX942-SDAG-NEXT:    v_mov_b32_e32 v2, v254
+; GFX942-SDAG-NEXT:    v_mov_b32_e32 v3, v255
+; GFX942-SDAG-NEXT:    v_mov_b32_e32 v4, v0
+; GFX942-SDAG-NEXT:    v_mov_b32_e32 v5, v1
+; GFX942-SDAG-NEXT:    s_set_gpr_idx_off
+; GFX942-SDAG-NEXT:    global_store_dwordx4 v[0:1], v[2:5], off
+; GFX942-SDAG-NEXT:    s_waitcnt vmcnt(0)
+; GFX942-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX942-GISEL-LABEL: load_straddling_end:
+; GFX942-GISEL:       ; %bb.0:
+; GFX942-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX942-GISEL-NEXT:    s_lshl2_add_u32 s0, s0, 0x3f8
+; GFX942-GISEL-NEXT:    s_lshr_b32 s0, s0, 2
+; GFX942-GISEL-NEXT:    s_set_gpr_idx_on s0, gpr_idx(SRC0)
+; GFX942-GISEL-NEXT:    v_mov_b32_e32 v2, v0
+; GFX942-GISEL-NEXT:    v_mov_b32_e32 v3, v1
+; GFX942-GISEL-NEXT:    v_mov_b32_e32 v4, v2
+; GFX942-GISEL-NEXT:    v_mov_b32_e32 v5, v3
+; GFX942-GISEL-NEXT:    s_set_gpr_idx_off
+; GFX942-GISEL-NEXT:    global_store_dwordx4 v[0:1], v[2:5], off
+; GFX942-GISEL-NEXT:    s_waitcnt vmcnt(0)
+; GFX942-GISEL-NEXT:    s_setpc_b64 s[30:31]
   %s = shl nuw i32 %i, 2
   %a = add nuw i32 %s, 1016
   %p = inttoptr i32 %a to ptr addrspace(13)
@@ -113,3 +183,4 @@ define void @load_straddling_end(ptr addrspace(1) %out, i32 inreg %i) {
 }
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; GFX12: {{.*}}
+; GFX942: {{.*}}



More information about the llvm-commits mailing list