[llvm] [AMDGPU] Simplify true16 SGPR folding. NFCI. (PR #226082)

Jay Foad via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 02:10:35 PDT 2026


https://github.com/jayfoad created https://github.com/llvm/llvm-project/pull/226082

Rewrite part of the SGPR folding "hack" from #128929 using
getChannelFromSubReg to avoid relying on the exact numbering of subreg
indices.

Co-authored-by: Claude Opus 5 (1M context) <noreply at anthropic.com>


>From ec2dddf8771987c2c35a2f25529a6808be6da99d Mon Sep 17 00:00:00 2001
From: Jay Foad <jay.foad at amd.com>
Date: Thu, 24 Sep 2026 10:07:37 +0100
Subject: [PATCH] [AMDGPU] Simplify true16 SGPR folding. NFCI.

Rewrite part of the SGPR folding "hack" from #128929 using
getChannelFromSubReg to avoid relying on the exact numbering of subreg
indices.

Co-authored-by: Claude Opus 5 (1M context) <noreply at anthropic.com>
---
 llvm/lib/Target/AMDGPU/SIFoldOperands.cpp | 35 +++++---------------
 llvm/test/CodeGen/AMDGPU/true16-fold.mir  | 39 +++++++++++++++++++++++
 2 files changed, 47 insertions(+), 27 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp b/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
index 9166ff0f1e5b1..d7831ace7eabc 100644
--- a/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
@@ -1515,34 +1515,15 @@ bool SIFoldOperandsImpl::foldOperand(
       // Hack to allow 32-bit SGPRs to be folded into True16 instructions
       // Remove this if 16-bit SGPRs (i.e. SGPR_LO16) are added to the
       // VS_16RegClass
-      //
-      // Excerpt from AMDGPUGenRegisterInfoEnums.inc
-      // NoSubRegister, //0
-      // hi16, // 1
-      // lo16, // 2
-      // sub0, // 3
-      // ...
-      // sub1, // 11
-      // sub1_hi16, // 12
-      // sub1_lo16, // 13
-      static_assert(AMDGPU::sub1_hi16 == 12, "Subregister layout has changed");
       if (Size == 2 && TRI->isVGPR(*MRI, UseMI->getOperand(0).getReg()) &&
-          TRI->isSGPRReg(*MRI, UseReg)) {
-        // Produce the 32 bit subregister index to which the 16-bit subregister
-        // is aligned.
-        if (SubRegIdx > AMDGPU::sub1) {
-          LaneBitmask M = TRI->getSubRegIndexLaneMask(SubRegIdx);
-          M |= M.getLane(M.getHighestLane() - 1);
-          SmallVector<unsigned, 4> Indexes;
-          TRI->getCoveringSubRegIndexes(TRI->getRegClassForReg(*MRI, UseReg), M,
-                                        Indexes);
-          assert(Indexes.size() == 1 && "Expected one 32-bit subreg to cover");
-          SubRegIdx = Indexes[0];
-          // 32-bit registers do not have a sub0 index
-        } else if (TII->getOpSize(*UseMI, 1) == 4)
-          SubRegIdx = 0;
-        else
-          SubRegIdx = AMDGPU::sub0;
+          TRI->isSGPRReg(*MRI, UseReg) && SubRegIdx != AMDGPU::NoSubRegister) {
+        // SGPRs only have lo16 subregisters, so the value is in the low half
+        // of a 32-bit SGPR. Use that whole 32-bit SGPR instead.
+        unsigned Channel = TRI->getChannelFromSubReg(SubRegIdx);
+        const TargetRegisterClass *UseRC = TRI->getRegClassForReg(*MRI, UseReg);
+        SubRegIdx = TRI->getRegSizeInBits(*UseRC) == 32
+                        ? AMDGPU::NoSubRegister
+                        : SIRegisterInfo::getSubRegFromChannel(Channel);
       }
       UseMI->getOperand(1).setSubReg(SubRegIdx);
       UseMI->getOperand(1).setIsKill(false);
diff --git a/llvm/test/CodeGen/AMDGPU/true16-fold.mir b/llvm/test/CodeGen/AMDGPU/true16-fold.mir
index 29be6c55e6f30..8892b7f2ff10f 100644
--- a/llvm/test/CodeGen/AMDGPU/true16-fold.mir
+++ b/llvm/test/CodeGen/AMDGPU/true16-fold.mir
@@ -39,6 +39,45 @@ body:             |
     S_ENDPGM 0, implicit %4
 ...
 
+---
+name:            fold_16bit_subreg_3
+tracksRegLiveness: true
+registers:
+body:             |
+  bb.0.entry:
+    ; CHECK-LABEL: name: fold_16bit_subreg_3
+    ; CHECK: [[DEF:%[0-9]+]]:sgpr_128 = IMPLICIT_DEF
+    ; CHECK-NEXT: [[DEF1:%[0-9]+]]:vgpr_16 = IMPLICIT_DEF
+    ; CHECK-NEXT: [[V_CMP_EQ_F16_t16_e64_:%[0-9]+]]:sreg_32 = nofpexcept V_CMP_EQ_F16_t16_e64 0, killed [[DEF1]], 2, [[DEF]].sub3, 0, 0, implicit $mode, implicit $exec
+    ; CHECK-NEXT: S_ENDPGM 0, implicit [[V_CMP_EQ_F16_t16_e64_]]
+    %0:sgpr_128 = IMPLICIT_DEF
+    %1:sgpr_lo16 = COPY %0.sub3_lo16:sgpr_128
+    %2:vgpr_16 = COPY %1:sgpr_lo16
+    %3:vgpr_16 = IMPLICIT_DEF
+    %4:sreg_32 = nofpexcept V_CMP_EQ_F16_t16_e64 0, killed %3:vgpr_16, 2, killed %2:vgpr_16, 0, 0, implicit $mode, implicit $exec
+    S_ENDPGM 0, implicit %4
+...
+
+---
+name:            fold_16bit_no_subreg
+tracksRegLiveness: true
+registers:
+body:             |
+  bb.0.entry:
+    ; CHECK-LABEL: name: fold_16bit_no_subreg
+    ; CHECK: [[DEF:%[0-9]+]]:sgpr_lo16 = IMPLICIT_DEF
+    ; CHECK-NEXT: [[COPY:%[0-9]+]]:vgpr_16 = COPY [[DEF]]
+    ; CHECK-NEXT: [[DEF1:%[0-9]+]]:vgpr_16 = IMPLICIT_DEF
+    ; CHECK-NEXT: [[V_CMP_EQ_F16_t16_e64_:%[0-9]+]]:sreg_32 = nofpexcept V_CMP_EQ_F16_t16_e64 0, killed [[DEF1]], 2, killed [[COPY]], 0, 0, implicit $mode, implicit $exec
+    ; CHECK-NEXT: S_ENDPGM 0, implicit [[V_CMP_EQ_F16_t16_e64_]]
+    %0:sgpr_lo16 = IMPLICIT_DEF
+    %1:sgpr_lo16 = COPY %0:sgpr_lo16
+    %2:vgpr_16 = COPY %1:sgpr_lo16
+    %3:vgpr_16 = IMPLICIT_DEF
+    %4:sreg_32 = nofpexcept V_CMP_EQ_F16_t16_e64 0, killed %3:vgpr_16, 2, killed %2:vgpr_16, 0, 0, implicit $mode, implicit $exec
+    S_ENDPGM 0, implicit %4
+...
+
 ---
 name:            sgpr_lo16
 tracksRegLiveness: true



More information about the llvm-commits mailing list