[llvm] [AMDGPU] Use S_MOV_B64_IMM_PSEUDO when moving 64-bit VGPR const to SGPR (PR #200576)

Arseniy Obolenskiy via llvm-commits llvm-commits at lists.llvm.org
Sun May 31 21:53:54 PDT 2026


https://github.com/aobolensk updated https://github.com/llvm/llvm-project/pull/200576

>From 90c86a84cdb4f73a671e2b3ec0b0f9976fdf162d Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Sat, 30 May 2026 15:31:05 +0200
Subject: [PATCH 1/2] [AMDGPU] Use S_MOV_B64_IMM_PSEUDO when moving 64-bit VGPR
 const to SGPR

S_MOV_B64 only encodes a 32-bit literal, so rematerializing a non-inline 64-bit immediate through it silently dropped the high 32 bits
---
 llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp    |  8 +-
 .../AMDGPU/vgpr_constant64_to_sgpr.mir        | 85 ++++++++++++++++++-
 2 files changed, 91 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
index bcefc3982b4d1..20a1f47ccb8ec 100644
--- a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
@@ -884,7 +884,13 @@ bool SIFixSGPRCopies::tryMoveVGPRConstToSGPR(
   const TargetRegisterClass *SrcRC =
       MRI->getRegClass(MaybeVGPRConstMO.getReg());
   unsigned MoveSize = TRI->getRegSizeInBits(*SrcRC);
-  unsigned MoveOp = MoveSize == 64 ? AMDGPU::S_MOV_B64 : AMDGPU::S_MOV_B32;
+  unsigned MoveOp;
+  if (MoveSize == 32)
+    MoveOp = AMDGPU::S_MOV_B32;
+  else if (MoveSize == 64)
+    MoveOp = AMDGPU::S_MOV_B64_IMM_PSEUDO;
+  else
+    return false;
   BuildMI(*BlockToInsertTo, PointToInsertTo, DL, TII->get(MoveOp), DstReg)
       .add(*SrcConst);
   if (MRI->hasOneUse(MaybeVGPRConstMO.getReg()))
diff --git a/llvm/test/CodeGen/AMDGPU/vgpr_constant64_to_sgpr.mir b/llvm/test/CodeGen/AMDGPU/vgpr_constant64_to_sgpr.mir
index 4729d9d8e4648..5c6ddec8e56ea 100644
--- a/llvm/test/CodeGen/AMDGPU/vgpr_constant64_to_sgpr.mir
+++ b/llvm/test/CodeGen/AMDGPU/vgpr_constant64_to_sgpr.mir
@@ -11,8 +11,91 @@ body:             |
   bb.0:
 
     ; GCN-LABEL: name: test_64imm
-    ; GCN: $sgpr8_sgpr9 = S_MOV_B64 4607182418800017408
+    ; GCN: $sgpr8_sgpr9 = S_MOV_B64_IMM_PSEUDO 4607182418800017408
     %1 = V_MOV_B64_PSEUDO 4607182418800017408, implicit $exec
     $sgpr8_sgpr9 = COPY %1
 ...
 
+# A non-inline 64-bit immediate (both halves non-zero) feeding a uniform-result
+# PHI must be rewritten with S_MOV_B64_IMM_PSEUDO, not S_MOV_B64. S_MOV_B64 only
+# encodes a 32-bit literal, which would silently drop the high 32 bits.
+
+---
+name:            test_b64_imm_phi
+tracksRegLiveness: true
+body:             |
+  ; GCN-LABEL: name: test_b64_imm_phi
+  ; GCN: bb.0:
+  ; GCN-NEXT:   successors: %bb.3(0x80000000)
+  ; GCN-NEXT:   liveins: $sgpr8_sgpr9, $sgpr10
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT:   [[COPY:%[0-9]+]]:sreg_64 = COPY $sgpr8_sgpr9
+  ; GCN-NEXT:   [[COPY1:%[0-9]+]]:sreg_32 = COPY $sgpr10
+  ; GCN-NEXT:   S_BRANCH %bb.3
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT: bb.1:
+  ; GCN-NEXT:   successors: %bb.3(0x40000000), %bb.2(0x40000000)
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT:   [[PHI:%[0-9]+]]:sreg_64 = PHI %12, %bb.3, %4, %bb.4
+  ; GCN-NEXT:   S_CMP_LG_U32 1, 1, implicit-def $scc
+  ; GCN-NEXT:   S_CBRANCH_SCC1 %bb.3, implicit $scc
+  ; GCN-NEXT:   S_BRANCH %bb.2
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT: bb.2:
+  ; GCN-NEXT:   S_ENDPGM 0
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT: bb.3:
+  ; GCN-NEXT:   successors: %bb.1(0x40000000), %bb.4(0x40000000)
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT:   [[PHI1:%[0-9]+]]:sreg_64 = PHI [[COPY]], %bb.0, [[PHI]], %bb.1
+  ; GCN-NEXT:   [[S_MOV_B64_:%[0-9]+]]:sreg_64 = S_MOV_B64 0
+  ; GCN-NEXT:   S_CMP_EQ_U64 [[PHI1]], killed [[S_MOV_B64_]], implicit-def $scc
+  ; GCN-NEXT:   [[S_MOV_B:%[0-9]+]]:sreg_64 = S_MOV_B64_IMM_PSEUDO 1311768467463790320
+  ; GCN-NEXT:   S_CBRANCH_SCC1 %bb.1, implicit $scc
+  ; GCN-NEXT:   S_BRANCH %bb.4
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT: bb.4:
+  ; GCN-NEXT:   successors: %bb.1(0x80000000)
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT:   [[S_LSHR_B64_:%[0-9]+]]:sreg_64 = S_LSHR_B64 [[PHI1]], [[COPY1]], implicit-def dead $scc
+  ; GCN-NEXT:   [[COPY2:%[0-9]+]]:sreg_32 = COPY [[S_LSHR_B64_]].sub1
+  ; GCN-NEXT:   [[S_LSHR_B64_1:%[0-9]+]]:sreg_64 = S_LSHR_B64 [[PHI1]], killed [[COPY2]], implicit-def dead $scc
+  ; GCN-NEXT:   [[COPY3:%[0-9]+]]:sreg_32 = COPY [[S_LSHR_B64_1]].sub1
+  ; GCN-NEXT:   [[S_OR_B32_:%[0-9]+]]:sreg_32 = S_OR_B32 killed [[COPY3]], [[COPY1]], implicit-def dead $scc
+  ; GCN-NEXT:   [[S_LSHR_B64_2:%[0-9]+]]:sreg_64 = S_LSHR_B64 [[PHI1]], killed [[S_OR_B32_]], implicit-def dead $scc
+  ; GCN-NEXT:   S_BRANCH %bb.1
+  bb.0:
+    liveins: $sgpr8_sgpr9, $sgpr10
+
+    %1:sreg_64 = COPY $sgpr8_sgpr9
+    %3:sreg_32 = COPY $sgpr10
+    %2:vreg_64 = V_MOV_B64_PSEUDO 1311768467463790320, implicit $exec
+    S_BRANCH %bb.3
+
+  bb.1:
+    %4:sreg_64 = PHI %2:vreg_64, %bb.3, %5:sreg_64, %bb.4
+    S_CMP_LG_U32 1, 1, implicit-def $scc
+    S_CBRANCH_SCC1 %bb.3, implicit $scc
+    S_BRANCH %bb.2
+
+  bb.2:
+    S_ENDPGM 0
+
+  bb.3:
+    %6:sreg_64 = PHI %1:sreg_64, %bb.0, %4:sreg_64, %bb.1
+    %7:sreg_64 = S_MOV_B64 0
+    S_CMP_EQ_U64 %6, killed %7, implicit-def $scc
+    S_CBRANCH_SCC1 %bb.1, implicit $scc
+    S_BRANCH %bb.4
+
+  bb.4:
+    %8:sreg_64 = S_LSHR_B64 %6, %3, implicit-def dead $scc
+    %9:sreg_32 = COPY %8.sub1
+    %10:sreg_64 = S_LSHR_B64 %6, killed %9, implicit-def dead $scc
+    %11:sreg_32 = COPY %10.sub1
+    %12:sreg_32 = S_OR_B32 killed %11, %3, implicit-def dead $scc
+    %5:sreg_64 = S_LSHR_B64 %6, killed %12, implicit-def dead $scc
+    S_BRANCH %bb.1
+
+...
+

>From 3fecef474da3b0f585c1157eac8a40c64e1819d3 Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Mon, 1 Jun 2026 06:53:44 +0200
Subject: [PATCH 2/2] Address comment

---
 llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp | 9 ++-------
 1 file changed, 2 insertions(+), 7 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
index 20a1f47ccb8ec..ffe34102b0a4e 100644
--- a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
@@ -884,13 +884,8 @@ bool SIFixSGPRCopies::tryMoveVGPRConstToSGPR(
   const TargetRegisterClass *SrcRC =
       MRI->getRegClass(MaybeVGPRConstMO.getReg());
   unsigned MoveSize = TRI->getRegSizeInBits(*SrcRC);
-  unsigned MoveOp;
-  if (MoveSize == 32)
-    MoveOp = AMDGPU::S_MOV_B32;
-  else if (MoveSize == 64)
-    MoveOp = AMDGPU::S_MOV_B64_IMM_PSEUDO;
-  else
-    return false;
+  unsigned MoveOp =
+      MoveSize == 64 ? AMDGPU::S_MOV_B64_IMM_PSEUDO : AMDGPU::S_MOV_B32;
   BuildMI(*BlockToInsertTo, PointToInsertTo, DL, TII->get(MoveOp), DstReg)
       .add(*SrcConst);
   if (MRI->hasOneUse(MaybeVGPRConstMO.getReg()))



More information about the llvm-commits mailing list