[llvm-branch-commits] [llvm] [AMDGPU] Make AMDGPURewriteAGPRCopyMFMA aware of subreg reload (PR #174998)

via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Tue Sep 29 06:01:14 PDT 2026


https://github.com/easyonaadit updated https://github.com/llvm/llvm-project/pull/174998

>From c76c2540db5798729644a96fe9ce8e5f3abc81ef Mon Sep 17 00:00:00 2001
From: Christudasan Devadasan <Christudasan.Devadasan at amd.com>
Date: Wed, 7 Jan 2026 10:48:26 +0000
Subject: [PATCH 1/4] [AMDGPU] Make AMDGPURewriteAGPRCopyMFMA aware of subreg
 reload

AMDGPURewriteAGPRCopyMFMA pass is currently not subreg-aware.
In particular, the logic that optimizes spills into COPY
instructions assumes full register reloads. This becomes
problematic when the reload instruction partially restores
a tuple register. This patch introduces the necessary changes
to make this pass subreg-aware, for a future patch that
implements subreg reload during RA.
---
 .../include/llvm/CodeGen/TargetRegisterInfo.h |  3 ++
 llvm/lib/CodeGen/TargetRegisterInfo.cpp       | 10 +++++
 .../AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp      | 42 ++++++++++++++++++-
 3 files changed, 54 insertions(+), 1 deletion(-)

diff --git a/llvm/include/llvm/CodeGen/TargetRegisterInfo.h b/llvm/include/llvm/CodeGen/TargetRegisterInfo.h
index 6b0e3e1289c85..2e976294f8f24 100644
--- a/llvm/include/llvm/CodeGen/TargetRegisterInfo.h
+++ b/llvm/include/llvm/CodeGen/TargetRegisterInfo.h
@@ -249,6 +249,9 @@ class LLVM_ABI TargetRegisterInfo : public MCRegisterInfo {
     return SubRegIndexLaneMasks[SubIdx];
   }
 
+  /// Try to find a matching subreg from the given lanemask.
+  unsigned getSubRegIdxFromLaneMask(LaneBitmask LaneMask) const;
+
   /// Try to find one or more subregister indexes to cover \p LaneMask.
   ///
   /// If this is possible, returns true and appends the best matching set of
diff --git a/llvm/lib/CodeGen/TargetRegisterInfo.cpp b/llvm/lib/CodeGen/TargetRegisterInfo.cpp
index 3b29b0863dd6c..7c522f0bc804e 100644
--- a/llvm/lib/CodeGen/TargetRegisterInfo.cpp
+++ b/llvm/lib/CodeGen/TargetRegisterInfo.cpp
@@ -495,6 +495,16 @@ TargetRegisterInfo::getRegSizeInBits(Register Reg,
   return getRegSizeInBits(*RC);
 }
 
+unsigned
+TargetRegisterInfo::getSubRegIdxFromLaneMask(LaneBitmask LaneMask) const {
+  for (unsigned Idx = 1, E = getNumSubRegIndices(); Idx < E; ++Idx) {
+    if (getSubRegIndexLaneMask(Idx) == LaneMask)
+      return Idx;
+  }
+
+  return 0 /*NoSubRegister*/;
+}
+
 bool TargetRegisterInfo::getCoveringSubRegIndexes(
     const TargetRegisterClass *RC, LaneBitmask LaneMask,
     SmallVectorImpl<unsigned> &NeededIndexes) const {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp b/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
index 5e27f39072ebb..46b972604db68 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
@@ -119,6 +119,17 @@ class AMDGPURewriteAGPRCopyMFMAImpl {
   bool tryFoldCopiesToAGPR(Register VReg, MCRegister AssignedAGPR) const;
   bool tryFoldCopiesFromAGPR(Register VReg, MCRegister AssignedAGPR) const;
 
+  /// Derives the subregister index from a spill reload pseudo instruction by
+  /// constructing a lane mask that covers the reloaded portion and finding
+  /// the matching subregister.
+  ///
+  /// \p MI the spill reload pseudo instruction containing the offset and
+  /// spill size info
+  /// \p Reg the original virtual register being spilled (mostly a tuple
+  /// register)
+  /// \return the subregister index corresponding to the reload portion.
+  unsigned getSubRegFromReload(MachineInstr &MI, Register VReg) const;
+
   /// Replace spill instruction \p SpillMI which loads/stores from/to \p SpillFI
   /// with a COPY to the replacement register value \p VReg.
   void replaceSpillWithCopyToVReg(MachineInstr &SpillMI, int SpillFI,
@@ -439,6 +450,33 @@ bool AMDGPURewriteAGPRCopyMFMAImpl::tryFoldCopiesFromAGPR(
   return MadeChange;
 }
 
+unsigned
+AMDGPURewriteAGPRCopyMFMAImpl::getSubRegFromReload(MachineInstr &MI,
+                                                   Register Reg) const {
+  unsigned NumRegs = TRI.getRegSizeInBits(*MRI.getRegClass(Reg)) / 32;
+  unsigned SubReg = 0;
+  // SubReg accesses for the tuple registers are of interest here.
+  // Note: We don't support 16-bit subreg reloads. If that assuption is
+  // changed in the future, this function should be revised.
+  if (NumRegs == 1)
+    return SubReg;
+
+  unsigned NumSpilledRegs = TII.getNumSubRegsForSpillOp(MI);
+  // Skip if the entire tuple is reloaded.
+  if (NumRegs == NumSpilledRegs)
+    return SubReg;
+
+  // Construct the covering lanes for the reloaded portion.
+  unsigned SubRegIdx =
+      TII.getNamedOperand(MI, AMDGPU::OpName::offset)->getImm() / 4;
+  // Subreg lane masks are maintained in terms of regunits and each 32-bit
+  // register consists of two regunits.
+  uint64_t Lanes = (1ULL << NumSpilledRegs * 2) - 1;
+  LaneBitmask CoveringLanes = LaneBitmask(Lanes << SubRegIdx * 2);
+  SubReg = TRI.getSubRegIdxFromLaneMask(CoveringLanes);
+  return SubReg;
+}
+
 void AMDGPURewriteAGPRCopyMFMAImpl::replaceSpillWithCopyToVReg(
     MachineInstr &SpillMI, int SpillFI, Register VReg) const {
   const DebugLoc &DL = SpillMI.getDebugLoc();
@@ -448,9 +486,11 @@ void AMDGPURewriteAGPRCopyMFMAImpl::replaceSpillWithCopyToVReg(
     NewCopy = BuildMI(MBB, SpillMI, DL, TII.get(TargetOpcode::COPY), VReg)
                   .add(SpillMI.getOperand(0));
   } else {
+    // Identify the subregs if SpillMI is really a subreg-load.
+    unsigned SubReg = getSubRegFromReload(SpillMI, VReg);
     NewCopy = BuildMI(MBB, SpillMI, DL, TII.get(TargetOpcode::COPY))
                   .add(SpillMI.getOperand(0))
-                  .addReg(VReg);
+                  .addReg(VReg, 0, SubReg);
   }
 
   LIS.ReplaceMachineInstrInMaps(SpillMI, *NewCopy);

>From 68e1bc8afd8ba9de6a62977a18c41fa24ee25757 Mon Sep 17 00:00:00 2001
From: Christudasan Devadasan <Christudasan.Devadasan at amd.com>
Date: Mon, 12 Jan 2026 12:59:34 +0000
Subject: [PATCH 2/4] suggestions incorporated.

---
 llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp b/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
index 46b972604db68..878019f1967c7 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
@@ -454,9 +454,9 @@ unsigned
 AMDGPURewriteAGPRCopyMFMAImpl::getSubRegFromReload(MachineInstr &MI,
                                                    Register Reg) const {
   unsigned NumRegs = TRI.getRegSizeInBits(*MRI.getRegClass(Reg)) / 32;
-  unsigned SubReg = 0;
+  unsigned SubReg = AMDGPU::NoSubRegister;
   // SubReg accesses for the tuple registers are of interest here.
-  // Note: We don't support 16-bit subreg reloads. If that assuption is
+  // Note: We don't support 16-bit subreg reloads. If that assumption is
   // changed in the future, this function should be revised.
   if (NumRegs == 1)
     return SubReg;

>From b4700efcface1f6181aa0a5dfa06b6d93e8ca863 Mon Sep 17 00:00:00 2001
From: Aaditya <Aaditya.AlokDeshpande at amd.com>
Date: Fri, 31 Jul 2026 14:27:15 +0530
Subject: [PATCH 3/4] Fix Merge conflict from rebasing.

---
 llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp b/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
index 878019f1967c7..0c6d23ed1ea20 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
@@ -490,7 +490,7 @@ void AMDGPURewriteAGPRCopyMFMAImpl::replaceSpillWithCopyToVReg(
     unsigned SubReg = getSubRegFromReload(SpillMI, VReg);
     NewCopy = BuildMI(MBB, SpillMI, DL, TII.get(TargetOpcode::COPY))
                   .add(SpillMI.getOperand(0))
-                  .addReg(VReg, 0, SubReg);
+                  .addReg(VReg, RegState::NoFlags, SubReg);
   }
 
   LIS.ReplaceMachineInstrInMaps(SpillMI, *NewCopy);

>From df21a78d644e1954fa30c31dd09bf405858e003b Mon Sep 17 00:00:00 2001
From: Aaditya <Aaditya.AlokDeshpande at amd.com>
Date: Tue, 29 Sep 2026 17:12:57 +0530
Subject: [PATCH 4/4] Added generic function to get subreg idx from offset and
 size

---
 .../include/llvm/CodeGen/TargetRegisterInfo.h |  7 +++---
 llvm/lib/CodeGen/TargetRegisterInfo.cpp       | 22 ++++++++--------
 .../AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp      | 25 ++++++-------------
 3 files changed, 24 insertions(+), 30 deletions(-)

diff --git a/llvm/include/llvm/CodeGen/TargetRegisterInfo.h b/llvm/include/llvm/CodeGen/TargetRegisterInfo.h
index 2e976294f8f24..11f63c6b7db1e 100644
--- a/llvm/include/llvm/CodeGen/TargetRegisterInfo.h
+++ b/llvm/include/llvm/CodeGen/TargetRegisterInfo.h
@@ -240,6 +240,10 @@ class LLVM_ABI TargetRegisterInfo : public MCRegisterInfo {
   /// access sub-registers at different offsets), return -1.
   unsigned getSubRegIdxOffset(unsigned Idx) const;
 
+  /// Find a SubReg index for the given bit size and bit offset.
+  /// Returns 0(no-op sub-register) if no matching index exists.
+  unsigned getSubRegIdxFromOffsetSize(unsigned Offset, unsigned Size) const;
+
   /// Return a bitmask representing the parts of a register that are covered by
   /// SubIdx \see LaneBitmask.
   ///
@@ -249,9 +253,6 @@ class LLVM_ABI TargetRegisterInfo : public MCRegisterInfo {
     return SubRegIndexLaneMasks[SubIdx];
   }
 
-  /// Try to find a matching subreg from the given lanemask.
-  unsigned getSubRegIdxFromLaneMask(LaneBitmask LaneMask) const;
-
   /// Try to find one or more subregister indexes to cover \p LaneMask.
   ///
   /// If this is possible, returns true and appends the best matching set of
diff --git a/llvm/lib/CodeGen/TargetRegisterInfo.cpp b/llvm/lib/CodeGen/TargetRegisterInfo.cpp
index 7c522f0bc804e..2a49e7e7d11c8 100644
--- a/llvm/lib/CodeGen/TargetRegisterInfo.cpp
+++ b/llvm/lib/CodeGen/TargetRegisterInfo.cpp
@@ -495,16 +495,6 @@ TargetRegisterInfo::getRegSizeInBits(Register Reg,
   return getRegSizeInBits(*RC);
 }
 
-unsigned
-TargetRegisterInfo::getSubRegIdxFromLaneMask(LaneBitmask LaneMask) const {
-  for (unsigned Idx = 1, E = getNumSubRegIndices(); Idx < E; ++Idx) {
-    if (getSubRegIndexLaneMask(Idx) == LaneMask)
-      return Idx;
-  }
-
-  return 0 /*NoSubRegister*/;
-}
-
 bool TargetRegisterInfo::getCoveringSubRegIndexes(
     const TargetRegisterClass *RC, LaneBitmask LaneMask,
     SmallVectorImpl<unsigned> &NeededIndexes) const {
@@ -613,6 +603,18 @@ unsigned TargetRegisterInfo::getSubRegIdxOffset(unsigned Idx) const {
   return SubRegIdxRanges[HwMode * getNumSubRegIndices() + Idx].Offset;
 }
 
+unsigned TargetRegisterInfo::getSubRegIdxFromOffsetSize(unsigned Offset,
+                                                        unsigned Size) const {
+  unsigned NumIdx = getNumSubRegIndices();
+  unsigned Base = HwMode * NumIdx;
+  for (unsigned Idx = 1; Idx < NumIdx; Idx++) {
+    if (SubRegIdxRanges[Base + Idx].Offset == Offset &&
+        SubRegIdxRanges[Base + Idx].Size == Size)
+      return Idx;
+  }
+  return 0;
+}
+
 Register
 TargetRegisterInfo::lookThruCopyLike(Register SrcReg,
                                      const MachineRegisterInfo *MRI) const {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp b/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
index 0c6d23ed1ea20..9f66d0959ecd9 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
@@ -454,27 +454,18 @@ unsigned
 AMDGPURewriteAGPRCopyMFMAImpl::getSubRegFromReload(MachineInstr &MI,
                                                    Register Reg) const {
   unsigned NumRegs = TRI.getRegSizeInBits(*MRI.getRegClass(Reg)) / 32;
-  unsigned SubReg = AMDGPU::NoSubRegister;
+  unsigned NumSpilledRegs = TII.getNumSubRegsForSpillOp(MI);
   // SubReg accesses for the tuple registers are of interest here.
+  // Skip if the entire tuple is reloaded.
   // Note: We don't support 16-bit subreg reloads. If that assumption is
   // changed in the future, this function should be revised.
-  if (NumRegs == 1)
-    return SubReg;
+  if (NumRegs == 1 || NumRegs == NumSpilledRegs)
+    return AMDGPU::NoSubRegister;
 
-  unsigned NumSpilledRegs = TII.getNumSubRegsForSpillOp(MI);
-  // Skip if the entire tuple is reloaded.
-  if (NumRegs == NumSpilledRegs)
-    return SubReg;
-
-  // Construct the covering lanes for the reloaded portion.
-  unsigned SubRegIdx =
-      TII.getNamedOperand(MI, AMDGPU::OpName::offset)->getImm() / 4;
-  // Subreg lane masks are maintained in terms of regunits and each 32-bit
-  // register consists of two regunits.
-  uint64_t Lanes = (1ULL << NumSpilledRegs * 2) - 1;
-  LaneBitmask CoveringLanes = LaneBitmask(Lanes << SubRegIdx * 2);
-  SubReg = TRI.getSubRegIdxFromLaneMask(CoveringLanes);
-  return SubReg;
+  unsigned StackSlotBitOffset =
+      TII.getNamedOperand(MI, AMDGPU::OpName::offset)->getImm() * 8;
+  return TRI.getSubRegIdxFromOffsetSize(StackSlotBitOffset,
+                                        NumSpilledRegs * 32);
 }
 
 void AMDGPURewriteAGPRCopyMFMAImpl::replaceSpillWithCopyToVReg(



More information about the llvm-branch-commits mailing list