[llvm] [AMDGPU] Mark constant HW-reg reads and MBCNT cheap-as-move to avoid nonlocal CSE (PR #228500)
Nick Riasanovsky via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 09:01:34 PDT 2026
https://github.com/njriasan created https://github.com/llvm/llvm-project/pull/228500
MachineCSE may freely eliminates redundant S_GETREG_B32_const and V_MBCNT_LO/HI_U32_B32 reads across basic blocks, extending an SGPR/VGPR live range over the region between definition and use. Both instructions are single SALU/VOP ops with no side effects (IntrNoMem, non-convergent) that read stable hardware state — constant HW-register fields and lane position — so recomputing them at the use is strictly cheaper than holding the value live. On AMDGPU that excess pressure directly costs occupancy.
This marks S_GETREG_B32_const (isReMaterializable + isAsCheapAsAMove, matching the existing S_MOV/S_BREV pattern) and both V_MBCNT defs (isAsCheapAsAMove, they are already rematerializable). MachineCSE heuristic #1 then keeps only same-block and immediate-predecessor CSE for them, while the pressure check still permits cross-block CSE when it provably adds no pressure.
This is motivated by an investigation on the Nvidia side where we found TLX kernels extending live ranges due to ctaid.x being pulled out of the warpspec region and extending live ranges: https://github.com/llvm/llvm-project/pull/228303. We do not immediately have a motivating AMD producer that demands it, but marking hardware register reads as "cheap to recompute" seem fundamentally sound.
TODO: Performance benchmarking to verify.
>From 81e573e8853159160b38958f820dd2ceb657c9e4 Mon Sep 17 00:00:00 2001
From: Nick Riasanovsky <njriasan at meta.com>
Date: Fri, 2 Oct 2026 08:51:56 -0700
Subject: [PATCH] Add isAsCheapAsAMove annotations
---
llvm/lib/Target/AMDGPU/SOPInstructions.td | 1 +
llvm/lib/Target/AMDGPU/VOP2Instructions.td | 4 +-
.../test/CodeGen/AMDGPU/machine-cse-hwreg.mir | 99 +++++++++++++++++++
3 files changed, 102 insertions(+), 2 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/machine-cse-hwreg.mir
diff --git a/llvm/lib/Target/AMDGPU/SOPInstructions.td b/llvm/lib/Target/AMDGPU/SOPInstructions.td
index b336af3bcd01f8..49a6bec876eac8 100644
--- a/llvm/lib/Target/AMDGPU/SOPInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SOPInstructions.td
@@ -1190,6 +1190,7 @@ def S_GETREG_B32 : S_GETREG_B32_Pseudo<
// A version of the pseudo for reading hardware register fields that are
// known to remain the same during the course of the run. Has no side
// effects and doesn't read MODE.
+let isReMaterializable = 1, isAsCheapAsAMove = 1 in
def S_GETREG_B32_const : S_GETREG_B32_Pseudo;
let Defs = [MODE], Uses = [MODE] in {
diff --git a/llvm/lib/Target/AMDGPU/VOP2Instructions.td b/llvm/lib/Target/AMDGPU/VOP2Instructions.td
index 8e0b23ca2de500..fcab619e11e1d4 100644
--- a/llvm/lib/Target/AMDGPU/VOP2Instructions.td
+++ b/llvm/lib/Target/AMDGPU/VOP2Instructions.td
@@ -978,10 +978,10 @@ foreach vt = Reg32Types.types in {
let isReMaterializable = 1 in {
defm V_BFM_B32 : VOP2Inst <"v_bfm_b32", VOP_I32_I32_I32>;
defm V_BCNT_U32_B32 : VOP2Inst <"v_bcnt_u32_b32", VOP_I32_I32_I32, add_ctpop>;
-let IsNeverUniform = 1 in {
+let IsNeverUniform = 1, isAsCheapAsAMove = 1 in {
defm V_MBCNT_LO_U32_B32 : VOP2Inst <"v_mbcnt_lo_u32_b32", VOP_I32_I32_I32, int_amdgcn_mbcnt_lo>;
defm V_MBCNT_HI_U32_B32 : VOP2Inst <"v_mbcnt_hi_u32_b32", VOP_I32_I32_I32, int_amdgcn_mbcnt_hi>;
-} // End IsNeverUniform = 1
+} // End IsNeverUniform = 1, isAsCheapAsAMove = 1
defm V_LDEXP_F32 : VOP2Inst <"v_ldexp_f32", VOP_F32_F32_I32, any_fldexp>;
let ReadsModeReg = 0, mayRaiseFPException = 0, SubtargetPredicate = HasCvtPkNormVOP2Insts in {
diff --git a/llvm/test/CodeGen/AMDGPU/machine-cse-hwreg.mir b/llvm/test/CodeGen/AMDGPU/machine-cse-hwreg.mir
new file mode 100644
index 00000000000000..e1361eb5298972
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/machine-cse-hwreg.mir
@@ -0,0 +1,99 @@
+# RUN: llc -mtriple=amdgpu10.30 -run-pass=machine-cse -verify-machineinstrs %s -o - 2>&1 | FileCheck --check-prefix=GCN %s
+# RUN: llc -mtriple=amdgpu10.30 -passes=machine-cse -verify-machineinstrs %s -o - 2>&1 | FileCheck --check-prefix=GCN %s
+# Constant hardware-register reads are single SALU/VOP moves, so MachineCSE
+# must only eliminate them locally, not across blocks.
+
+# GCN-LABEL: name: keep_getreg_const
+# GCN-COUNT-2: S_GETREG_B32_const
+---
+name: keep_getreg_const
+body: |
+ bb.0:
+ successors: %bb.1(0x80000000)
+ %0:sreg_32 = S_GETREG_B32_const 1234
+ S_NOP 0, implicit %0
+ S_BRANCH %bb.1
+
+ bb.1:
+ successors: %bb.2(0x80000000)
+ S_NOP 0
+ S_BRANCH %bb.2
+
+ bb.2:
+ %1:sreg_32 = S_GETREG_B32_const 1234
+ S_NOP 0, implicit %1
+ S_ENDPGM 0
+...
+
+# GCN-LABEL: name: keep_mbcnt_lo
+# GCN-COUNT-2: V_MBCNT_LO_U32_B32_e32
+---
+name: keep_mbcnt_lo
+body: |
+ bb.0:
+ successors: %bb.1(0x80000000)
+ %base:vgpr_32 = IMPLICIT_DEF
+ %0:vgpr_32 = V_MBCNT_LO_U32_B32_e32 -1, %base, implicit $exec
+ S_NOP 0, implicit %0
+ S_BRANCH %bb.1
+
+ bb.1:
+ successors: %bb.2(0x80000000)
+ S_NOP 0
+ S_BRANCH %bb.2
+
+ bb.2:
+ %1:vgpr_32 = V_MBCNT_LO_U32_B32_e32 -1, %base, implicit $exec
+ S_NOP 0, implicit %1
+ S_ENDPGM 0
+...
+
+# GCN-LABEL: name: keep_mbcnt_hi
+# GCN-COUNT-2: V_MBCNT_HI_U32_B32_e32
+---
+name: keep_mbcnt_hi
+body: |
+ bb.0:
+ successors: %bb.1(0x80000000)
+ %base:vgpr_32 = IMPLICIT_DEF
+ %0:vgpr_32 = V_MBCNT_HI_U32_B32_e32 -1, %base, implicit $exec
+ S_NOP 0, implicit %0
+ S_BRANCH %bb.1
+
+ bb.1:
+ successors: %bb.2(0x80000000)
+ S_NOP 0
+ S_BRANCH %bb.2
+
+ bb.2:
+ %1:vgpr_32 = V_MBCNT_HI_U32_B32_e32 -1, %base, implicit $exec
+ S_NOP 0, implicit %1
+ S_ENDPGM 0
+...
+
+# GCN-LABEL: name: cse_getreg_const_locally
+# GCN-COUNT-1: S_GETREG_B32_const
+---
+name: cse_getreg_const_locally
+body: |
+ bb.0:
+ %0:sreg_32 = S_GETREG_B32_const 1234
+ %1:sreg_32 = S_GETREG_B32_const 1234
+ S_NOP 0, implicit %0
+ S_NOP 0, implicit %1
+ S_ENDPGM 0
+...
+
+# GCN-LABEL: name: cse_mbcnt_lo_locally
+# GCN-COUNT-1: V_MBCNT_LO_U32_B32_e32
+---
+name: cse_mbcnt_lo_locally
+body: |
+ bb.0:
+ %base:vgpr_32 = IMPLICIT_DEF
+ %0:vgpr_32 = V_MBCNT_LO_U32_B32_e32 -1, %base, implicit $exec
+ %1:vgpr_32 = V_MBCNT_LO_U32_B32_e32 -1, %base, implicit $exec
+ S_NOP 0, implicit %0
+ S_NOP 0, implicit %1
+ S_ENDPGM 0
+...
More information about the llvm-commits
mailing list