[llvm] [Draft][AMDGPU] Take into account amdgpu-waves-per-eu in getRegPressureLimit (PR #173992)
Juan Manuel Martinez CaamaƱo via llvm-commits
llvm-commits at lists.llvm.org
Tue Dec 30 06:41:30 PST 2025
https://github.com/jmmartinez created https://github.com/llvm/llvm-project/pull/173992
**!!! At the moment the tests fail because this patch depends on https://github.com/llvm/llvm-project/pull/168358 !!!**
I didn't rebase it over it to keep a smaller diff.
The minimum occupancy computed by `getOccupancyWithWorkGroupSizes`
doesn't take into account that the user may have provided a
low-occupancy target through the `amdgpu-waves-per-eu` attribute.
Bound the minimum-occupancy using the maximum value in
`amdgpu-waves-per-eu`.
When the user specifies a small `amdgpu-waves-per-eu` (like `"1,1"`), this
results in higher vpgr limits.
>From 73075194a669c53a0ec63df372e189891e2d7f00 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Juan=20Manuel=20Martinez=20Caama=C3=B1o?=
<jmartinezcaamao at gmail.com>
Date: Tue, 30 Dec 2025 14:01:37 +0100
Subject: [PATCH 1/2] Pre-commit test: [AMDGPU] Take into account
amdgpu-waves-per-eu in getRegPressureLimit
---
llvm/test/CodeGen/AMDGPU/licm-regpressure.mir | 156 +++++++++++++++++-
1 file changed, 154 insertions(+), 2 deletions(-)
diff --git a/llvm/test/CodeGen/AMDGPU/licm-regpressure.mir b/llvm/test/CodeGen/AMDGPU/licm-regpressure.mir
index 98552de05c857..d2886ff9ee448 100644
--- a/llvm/test/CodeGen/AMDGPU/licm-regpressure.mir
+++ b/llvm/test/CodeGen/AMDGPU/licm-regpressure.mir
@@ -4,12 +4,21 @@
# MachineLICM shall limit hoisting of V_CVT instructions out of the loop keeping
# register pressure within the budget. VGPR budget at occupancy 10 is 24 vgprs.
+# Respect this limit, unless the user explicitely limited the occupancy.
+# Both tests have the same mir, only the amdgpu-waves-per-eu attribute changes.
+
+--- |
+ define amdgpu_kernel void @test_default() { ret void }
+
+ define amdgpu_kernel void @test_eu_wave_limit() #0 { ret void }
+
+ attributes #0 = { "amdgpu-waves-per-eu"="1,1" }
---
-name: test
+name: test_default
tracksRegLiveness: true
body: |
- ; GCN-LABEL: name: test
+ ; GCN-LABEL: name: test_default
; GCN: bb.0:
; GCN-NEXT: successors: %bb.1(0x80000000)
; GCN-NEXT: liveins: $vcc, $vgpr0
@@ -148,5 +157,148 @@ body: |
bb.2:
S_ENDPGM 0
+...
+---
+name: test_eu_wave_limit
+tracksRegLiveness: true
+body: |
+ ; GCN-LABEL: name: test_eu_wave_limit
+ ; GCN: bb.0:
+ ; GCN-NEXT: successors: %bb.1(0x80000000)
+ ; GCN-NEXT: liveins: $vcc, $vgpr0
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: [[COPY:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY1:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY2:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY3:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY4:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY5:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY6:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY7:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY8:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY9:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY10:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY11:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY12:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY13:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY14:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY15:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY16:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[COPY17:%[0-9]+]]:vgpr_32 = COPY $vgpr0
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_1:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY1]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_2:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY2]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_3:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY3]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_4:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY4]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_5:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY5]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_6:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY6]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_7:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY7]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_8:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY8]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_9:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY9]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_10:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY10]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_11:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY11]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_12:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY12]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_13:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY13]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_14:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY14]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_15:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY15]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_16:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY16]], implicit $mode, implicit $exec
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.1:
+ ; GCN-NEXT: successors: %bb.2(0x04000000), %bb.1(0x7c000000)
+ ; GCN-NEXT: liveins: $vcc
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: $vcc = S_AND_B64 $exec, $vcc, implicit-def $scc
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_1]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_2]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_3]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_4]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_5]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_6]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_7]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_8]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_9]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_10]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_11]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_12]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_13]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_14]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_15]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_16]], implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_17:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY17]], implicit $mode, implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, killed [[V_CVT_F64_I32_e32_17]], implicit $exec
+ ; GCN-NEXT: S_CBRANCH_VCCNZ %bb.1, implicit $vcc
+ ; GCN-NEXT: S_BRANCH %bb.2
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.2:
+ ; GCN-NEXT: S_ENDPGM 0
+ bb.0:
+ successors: %bb.1(0x80000000)
+ liveins: $vcc, $vgpr0
+
+ %0:vgpr_32 = COPY $vgpr0
+ %1:vgpr_32 = COPY $vgpr0
+ %2:vgpr_32 = COPY $vgpr0
+ %3:vgpr_32 = COPY $vgpr0
+ %4:vgpr_32 = COPY $vgpr0
+ %5:vgpr_32 = COPY $vgpr0
+ %6:vgpr_32 = COPY $vgpr0
+ %7:vgpr_32 = COPY $vgpr0
+ %8:vgpr_32 = COPY $vgpr0
+ %9:vgpr_32 = COPY $vgpr0
+ %10:vgpr_32 = COPY $vgpr0
+ %11:vgpr_32 = COPY $vgpr0
+ %12:vgpr_32 = COPY $vgpr0
+ %13:vgpr_32 = COPY $vgpr0
+ %14:vgpr_32 = COPY $vgpr0
+ %15:vgpr_32 = COPY $vgpr0
+ %16:vgpr_32 = COPY $vgpr0
+ %17:vgpr_32 = COPY $vgpr0
+
+ bb.1:
+ successors: %bb.2(0x04000000), %bb.1(0x7c000000)
+ liveins: $vcc
+
+ $vcc = S_AND_B64 $exec, $vcc, implicit-def $scc
+ %18:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %0, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %18, implicit $exec
+ %19:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %1, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %19, implicit $exec
+ %20:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %2, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %20, implicit $exec
+ %21:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %3, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %21, implicit $exec
+ %22:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %4, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %22, implicit $exec
+ %23:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %5, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %23, implicit $exec
+ %24:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %6, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %24, implicit $exec
+ %25:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %7, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %25, implicit $exec
+ %26:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %8, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %26, implicit $exec
+ %27:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %9, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %27, implicit $exec
+ %28:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %10, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %28, implicit $exec
+ %29:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %11, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %29, implicit $exec
+ %30:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %12, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %30, implicit $exec
+ %31:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %13, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %31, implicit $exec
+ %32:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %14, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %32, implicit $exec
+ %33:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %15, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %33, implicit $exec
+ %34:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %16, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %34, implicit $exec
+ %35:vreg_64 = nofpexcept V_CVT_F64_I32_e32 %17, implicit $mode, implicit $exec
+ $vcc = V_CMP_EQ_U64_e64 $vcc, killed %35, implicit $exec
+ S_CBRANCH_VCCNZ %bb.1, implicit $vcc
+ S_BRANCH %bb.2
+ bb.2:
+ S_ENDPGM 0
...
>From 74a8de50ed9b237008fa056559426c340ea79ea8 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Juan=20Manuel=20Martinez=20Caama=C3=B1o?=
<jmartinezcaamao at gmail.com>
Date: Tue, 30 Dec 2025 11:11:27 +0100
Subject: [PATCH 2/2] [AMDGPU] Take into account amdgpu-waves-per-eu in
getRegPressureLimit
The minimum occupancy computed by `getOccupancyWithWorkGroupSizes`
doesn't take into account that the user may have provided a
low-occupancy target through the amdgpu-waves-per-eu attribute.
Bound the minimum-occupancy using the maximum value in
amdgpu-waves-per-eu.
When the user specifies a small amdgpu-waves-per-eu (like "1,1"), this
results in higher vpgr limits.
This patch depends on https://github.com/llvm/llvm-project/pull/168358
to work.
---
llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp | 8 ++-
.../AMDGPU/agpr-copy-no-free-registers.ll | 70 +++++++------------
llvm/test/CodeGen/AMDGPU/licm-regpressure.mir | 4 +-
3 files changed, 36 insertions(+), 46 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp b/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
index 66586e8bc234a..ac3589e4098c4 100644
--- a/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
@@ -3783,7 +3783,13 @@ bool SIRegisterInfo::isAGPR(const MachineRegisterInfo &MRI,
unsigned SIRegisterInfo::getRegPressureLimit(const TargetRegisterClass *RC,
MachineFunction &MF) const {
- unsigned MinOcc = ST.getOccupancyWithWorkGroupSizes(MF).first;
+ unsigned MinWGSizeOcc = ST.getOccupancyWithWorkGroupSizes(MF).first;
+ unsigned MaxWavesPerEU = ST.getWavesPerEU(MF.getFunction()).second;
+ // The occupancy is bound by the amdgpu-waves-per-eu value specified by the
+ // user. Using a small value for amdgpu-waves-per-eu, should yield bigger
+ // register pressure limits (if resources allow it).
+ unsigned MinOcc = std::min(MinWGSizeOcc, MaxWavesPerEU);
+
switch (RC->getID()) {
default:
return AMDGPUGenRegisterInfo::getRegPressureLimit(RC, MF);
diff --git a/llvm/test/CodeGen/AMDGPU/agpr-copy-no-free-registers.ll b/llvm/test/CodeGen/AMDGPU/agpr-copy-no-free-registers.ll
index ef7a13819a799..50bbae1fdcd99 100644
--- a/llvm/test/CodeGen/AMDGPU/agpr-copy-no-free-registers.ll
+++ b/llvm/test/CodeGen/AMDGPU/agpr-copy-no-free-registers.ll
@@ -375,64 +375,48 @@ define void @v32_asm_def_use(float %v0, float %v1) #4 {
; GFX908-NEXT: ;;#ASMSTART
; GFX908-NEXT: ; def v[0:31] a[0:15]
; GFX908-NEXT: ;;#ASMEND
-; GFX908-NEXT: v_accvgpr_read_b32 v35, a15
-; GFX908-NEXT: ;;#ASMSTART
-; GFX908-NEXT: ; def v32
-; GFX908-NEXT: ;;#ASMEND
-; GFX908-NEXT: s_nop 1
-; GFX908-NEXT: v_accvgpr_write_b32 a31, v35
+; GFX908-NEXT: v_accvgpr_read_b32 v32, a15
; GFX908-NEXT: v_accvgpr_read_b32 v35, a14
-; GFX908-NEXT: s_nop 1
+; GFX908-NEXT: v_accvgpr_read_b32 v36, a13
+; GFX908-NEXT: v_accvgpr_write_b32 a31, v32
+; GFX908-NEXT: v_accvgpr_read_b32 v32, a12
; GFX908-NEXT: v_accvgpr_write_b32 a30, v35
-; GFX908-NEXT: v_accvgpr_read_b32 v35, a13
-; GFX908-NEXT: s_nop 1
-; GFX908-NEXT: v_accvgpr_write_b32 a29, v35
-; GFX908-NEXT: v_accvgpr_read_b32 v35, a12
-; GFX908-NEXT: s_nop 1
-; GFX908-NEXT: v_accvgpr_write_b32 a28, v35
+; GFX908-NEXT: v_accvgpr_write_b32 a29, v36
+; GFX908-NEXT: v_accvgpr_write_b32 a28, v32
; GFX908-NEXT: v_accvgpr_read_b32 v35, a11
-; GFX908-NEXT: s_nop 1
+; GFX908-NEXT: v_accvgpr_read_b32 v36, a10
+; GFX908-NEXT: v_accvgpr_read_b32 v32, a9
; GFX908-NEXT: v_accvgpr_write_b32 a27, v35
-; GFX908-NEXT: v_accvgpr_read_b32 v35, a10
-; GFX908-NEXT: s_nop 1
-; GFX908-NEXT: v_accvgpr_write_b32 a26, v35
-; GFX908-NEXT: v_accvgpr_read_b32 v35, a9
-; GFX908-NEXT: s_nop 1
-; GFX908-NEXT: v_accvgpr_write_b32 a25, v35
+; GFX908-NEXT: v_accvgpr_write_b32 a26, v36
+; GFX908-NEXT: v_accvgpr_write_b32 a25, v32
; GFX908-NEXT: v_accvgpr_read_b32 v35, a8
-; GFX908-NEXT: s_nop 1
+; GFX908-NEXT: v_accvgpr_read_b32 v36, a7
+; GFX908-NEXT: v_accvgpr_read_b32 v32, a6
; GFX908-NEXT: v_accvgpr_write_b32 a24, v35
-; GFX908-NEXT: v_accvgpr_read_b32 v35, a7
-; GFX908-NEXT: s_nop 1
-; GFX908-NEXT: v_accvgpr_write_b32 a23, v35
-; GFX908-NEXT: v_accvgpr_read_b32 v35, a6
-; GFX908-NEXT: s_nop 1
-; GFX908-NEXT: v_accvgpr_write_b32 a22, v35
+; GFX908-NEXT: v_accvgpr_write_b32 a23, v36
+; GFX908-NEXT: v_accvgpr_write_b32 a22, v32
; GFX908-NEXT: v_accvgpr_read_b32 v35, a5
-; GFX908-NEXT: s_nop 1
+; GFX908-NEXT: v_accvgpr_read_b32 v36, a4
+; GFX908-NEXT: v_accvgpr_read_b32 v32, a3
; GFX908-NEXT: v_accvgpr_write_b32 a21, v35
-; GFX908-NEXT: v_accvgpr_read_b32 v35, a4
-; GFX908-NEXT: s_nop 1
-; GFX908-NEXT: v_accvgpr_write_b32 a20, v35
-; GFX908-NEXT: v_accvgpr_read_b32 v35, a3
-; GFX908-NEXT: s_nop 1
-; GFX908-NEXT: v_accvgpr_write_b32 a19, v35
+; GFX908-NEXT: v_accvgpr_write_b32 a20, v36
+; GFX908-NEXT: v_accvgpr_write_b32 a19, v32
; GFX908-NEXT: v_accvgpr_read_b32 v35, a2
-; GFX908-NEXT: s_nop 1
+; GFX908-NEXT: v_accvgpr_read_b32 v36, a1
+; GFX908-NEXT: v_accvgpr_read_b32 v32, a0
; GFX908-NEXT: v_accvgpr_write_b32 a18, v35
-; GFX908-NEXT: v_accvgpr_read_b32 v35, a1
-; GFX908-NEXT: s_nop 1
-; GFX908-NEXT: v_accvgpr_write_b32 a17, v35
-; GFX908-NEXT: v_accvgpr_read_b32 v35, a0
-; GFX908-NEXT: s_nop 1
-; GFX908-NEXT: v_accvgpr_write_b32 a16, v35
+; GFX908-NEXT: v_accvgpr_write_b32 a17, v36
+; GFX908-NEXT: v_accvgpr_write_b32 a16, v32
+; GFX908-NEXT: ;;#ASMSTART
+; GFX908-NEXT: ; def v32
+; GFX908-NEXT: ;;#ASMEND
; GFX908-NEXT: ;;#ASMSTART
; GFX908-NEXT: ; copy
; GFX908-NEXT: ;;#ASMEND
-; GFX908-NEXT: v_accvgpr_read_b32 v35, a1
+; GFX908-NEXT: v_accvgpr_read_b32 v37, a1
; GFX908-NEXT: v_mfma_f32_16x16x1f32 a[0:15], v34, v33, a[16:31]
; GFX908-NEXT: s_nop 0
-; GFX908-NEXT: v_accvgpr_write_b32 a32, v35
+; GFX908-NEXT: v_accvgpr_write_b32 a32, v37
; GFX908-NEXT: ;;#ASMSTART
; GFX908-NEXT: ; copy
; GFX908-NEXT: ;;#ASMEND
diff --git a/llvm/test/CodeGen/AMDGPU/licm-regpressure.mir b/llvm/test/CodeGen/AMDGPU/licm-regpressure.mir
index d2886ff9ee448..78025ea0c2978 100644
--- a/llvm/test/CodeGen/AMDGPU/licm-regpressure.mir
+++ b/llvm/test/CodeGen/AMDGPU/licm-regpressure.mir
@@ -202,6 +202,7 @@ body: |
; GCN-NEXT: [[V_CVT_F64_I32_e32_14:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY14]], implicit $mode, implicit $exec
; GCN-NEXT: [[V_CVT_F64_I32_e32_15:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY15]], implicit $mode, implicit $exec
; GCN-NEXT: [[V_CVT_F64_I32_e32_16:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY16]], implicit $mode, implicit $exec
+ ; GCN-NEXT: [[V_CVT_F64_I32_e32_17:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY17]], implicit $mode, implicit $exec
; GCN-NEXT: {{ $}}
; GCN-NEXT: bb.1:
; GCN-NEXT: successors: %bb.2(0x04000000), %bb.1(0x7c000000)
@@ -225,8 +226,7 @@ body: |
; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_14]], implicit $exec
; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_15]], implicit $exec
; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_16]], implicit $exec
- ; GCN-NEXT: [[V_CVT_F64_I32_e32_17:%[0-9]+]]:vreg_64 = nofpexcept V_CVT_F64_I32_e32 [[COPY17]], implicit $mode, implicit $exec
- ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, killed [[V_CVT_F64_I32_e32_17]], implicit $exec
+ ; GCN-NEXT: $vcc = V_CMP_EQ_U64_e64 $vcc, [[V_CVT_F64_I32_e32_17]], implicit $exec
; GCN-NEXT: S_CBRANCH_VCCNZ %bb.1, implicit $vcc
; GCN-NEXT: S_BRANCH %bb.2
; GCN-NEXT: {{ $}}
More information about the llvm-commits
mailing list