[llvm-branch-commits] [llvm] [AMDGPU][SIMemoryLegalizer] Lower single-wave workgroup scope to wavefront in absence of LDSDMA (PR #224065)
Zach Goldthorpe via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Mon Sep 21 07:54:25 PDT 2026
https://github.com/zGoldthorpe updated https://github.com/llvm/llvm-project/pull/224065
>From 8bc6201f64e3e3e80a296f6c77de4de9fc4d640b Mon Sep 17 00:00:00 2001
From: Zach Goldthorpe <Zach.Goldthorpe at amd.com>
Date: Wed, 16 Sep 2026 09:42:10 -0500
Subject: [PATCH 1/4] Reapply "[AMDGPU] Use wavefront scope for single-wave
workgroup synchronization (#187673)"
---
llvm/docs/AMDGPUUsage.rst | 12 +
.../Target/AMDGPU/AMDGPULowerIntrinsics.cpp | 6 +-
llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp | 4 +
llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h | 4 +
llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp | 32 +-
.../test/CodeGen/AMDGPU/flat-saddr-atomics.ll | 192 ++-
.../CodeGen/AMDGPU/global-saddr-atomics.ll | 672 +++-------
llvm/test/CodeGen/AMDGPU/lds-dma-war-async.ll | 6 +-
llvm/test/CodeGen/AMDGPU/lds-dma-war.ll | 7 +-
...zer-single-wave-workgroup-lds-dma-async.ll | 6 -
...legalizer-single-wave-workgroup-lds-dma.ll | 64 +-
...-legalizer-single-wave-workgroup-memops.ll | 1123 ++++++++---------
12 files changed, 885 insertions(+), 1243 deletions(-)
diff --git a/llvm/docs/AMDGPUUsage.rst b/llvm/docs/AMDGPUUsage.rst
index b0af2849d94a86..dedbec9c6a415c 100644
--- a/llvm/docs/AMDGPUUsage.rst
+++ b/llvm/docs/AMDGPUUsage.rst
@@ -7545,6 +7545,18 @@ treated as non-atomic.
A memory synchronization scope wider than work-group is not meaningful for the
group (LDS) address space and is treated as work-group.
+When a work-group's maximum flat work-group size does not exceed the wavefront
+size, the work-group fits within a single wavefront. In this case, LLVM
+``workgroup`` synchronization scope is equivalent to ``wavefront`` scope.
+
+If the compiler can determine this bound (e.g., via ``amdgpu-flat-work-group-size``),
+the AMDGPU backend optimizes ``workgroup`` scope operations by lowering them to
+``wavefront``-scoped machine instructions.
+
+It applies to atomic ``load``, ``store``, ``atomicrmw``, and ``cmpxchg``
+instructions, and to ``fence`` instructions, when they use synchronizing memory
+orderings (``acquire``, ``release``, ``acq_rel``, or ``seq_cst``).
+
The memory model does not support the region address space which is treated as
non-atomic.
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerIntrinsics.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerIntrinsics.cpp
index 6150e14f38c7df..83ea136ab9fef9 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerIntrinsics.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerIntrinsics.cpp
@@ -105,10 +105,8 @@ bool AMDGPULowerIntrinsicsImpl::visitBarrier(IntrinsicInst &I) {
const GCNSubtarget &ST = TM.getSubtarget<GCNSubtarget>(*I.getFunction());
bool IsSingleWaveWG = false;
- if (TM.getOptLevel() > CodeGenOptLevel::None) {
- unsigned WGMaxSize = ST.getFlatWorkGroupSizes(*I.getFunction()).second;
- IsSingleWaveWG = WGMaxSize <= ST.getWavefrontSize();
- }
+ if (TM.getOptLevel() > CodeGenOptLevel::None)
+ IsSingleWaveWG = ST.isSingleWavefrontWorkgroup(*I.getFunction());
IRBuilder<> B(&I);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp
index b9ba99cf8db19c..ca78fa59110842 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp
@@ -177,6 +177,10 @@ std::pair<unsigned, unsigned> AMDGPUSubtarget::getFlatWorkGroupSizes(
return Requested;
}
+bool AMDGPUSubtarget::isSingleWavefrontWorkgroup(const Function &F) const {
+ return getFlatWorkGroupSizes(F).second <= getWavefrontSize();
+}
+
std::pair<unsigned, unsigned> AMDGPUSubtarget::getEffectiveWavesPerEU(
std::pair<unsigned, unsigned> RequestedWavesPerEU,
std::pair<unsigned, unsigned> FlatWorkGroupSizes, unsigned LDSBytes) const {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
index e1f331229c463b..4fa6020e9198ce 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
@@ -82,6 +82,10 @@ class AMDGPUSubtarget {
/// be converted to integer, or violate subtarget's specifications.
std::pair<unsigned, unsigned> getFlatWorkGroupSizes(const Function &F) const;
+ /// \returns true if the maximum flat work-group size for \p F is at most the
+ /// wavefront size, so a work-group may fit in a single wavefront.
+ bool isSingleWavefrontWorkgroup(const Function &F) const;
+
/// \returns The required size of workgroups that will be used to execute \p F
/// in the \p Dim dimension, if it is known (from `!reqd_work_group_size`
/// metadata. Otherwise, returns std::nullopt.
diff --git a/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp b/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
index e818e06e3bb048..b6884f2a51268b 100644
--- a/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
+++ b/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
@@ -159,7 +159,8 @@ class SIMemOpInfo final {
bool IsCrossAddressSpaceOrdering = true,
AtomicOrdering FailureOrdering = AtomicOrdering::SequentiallyConsistent,
bool IsVolatile = false, bool IsNonTemporal = false,
- bool IsLastUse = false, bool IsCooperative = false, bool IsAVNone = false)
+ bool IsLastUse = false, bool IsCooperative = false, bool IsAVNone = false,
+ bool CanDemoteWorkgroupToWavefront = false)
: Ordering(Ordering), FailureOrdering(FailureOrdering), Scope(Scope),
OrderingAddrSpace(OrderingAddrSpace), InstrAddrSpace(InstrAddrSpace),
IsCrossAddressSpaceOrdering(IsCrossAddressSpaceOrdering),
@@ -207,6 +208,17 @@ class SIMemOpInfo final {
// AGENT scope as a conservatively correct alternative.
if (this->Scope == SIAtomicScope::CLUSTER && !ST.hasClusters())
this->Scope = SIAtomicScope::AGENT;
+
+ // When max flat work-group size is at most the wavefront size, the
+ // work-group fits in a single wave, so LLVM workgroup scope matches
+ // wavefront scope. Demote workgroup → wavefront here for fences and for
+ // atomics with ordering stronger than monotonic.
+ if (CanDemoteWorkgroupToWavefront &&
+ this->Scope == SIAtomicScope::WORKGROUP &&
+ (llvm::isStrongerThan(this->Ordering, AtomicOrdering::Monotonic) ||
+ llvm::isStrongerThan(this->FailureOrdering,
+ AtomicOrdering::Monotonic)))
+ this->Scope = SIAtomicScope::WAVEFRONT;
}
public:
@@ -280,6 +292,7 @@ class SIMemOpAccess final {
private:
const AMDGPUMachineModuleInfo *MMI = nullptr;
const GCNSubtarget &ST;
+ const bool CanDemoteWorkgroupToWavefront;
/// Reports unsupported message \p Msg for \p MI to LLVM context.
void reportUnsupported(const MachineBasicBlock::iterator &MI,
@@ -303,7 +316,8 @@ class SIMemOpAccess final {
public:
/// Construct class to support accessing the machine memory operands
/// of instructions.
- SIMemOpAccess(const AMDGPUMachineModuleInfo &MMI, const GCNSubtarget &ST);
+ SIMemOpAccess(const AMDGPUMachineModuleInfo &MMI, const GCNSubtarget &ST,
+ const Function &F);
/// \returns Load info if \p MI is a load operation, "std::nullopt" otherwise.
std::optional<SIMemOpInfo>
@@ -824,9 +838,13 @@ SIAtomicAddrSpace SIMemOpAccess::toSIAtomicAddrSpace(unsigned AS) const {
return SIAtomicAddrSpace::OTHER;
}
+// TODO: Consider moving single-wave workgroup->wavefront scope relaxation to an
+// IR pass (and extending it to other scoped operations), so middle-end
+// optimizations see wavefront scope earlier.
SIMemOpAccess::SIMemOpAccess(const AMDGPUMachineModuleInfo &MMI_,
- const GCNSubtarget &ST)
- : MMI(&MMI_), ST(ST) {}
+ const GCNSubtarget &ST, const Function &F)
+ : MMI(&MMI_), ST(ST),
+ CanDemoteWorkgroupToWavefront(ST.isSingleWavefrontWorkgroup(F)) {}
std::optional<SIMemOpInfo> SIMemOpAccess::constructFromMIWithMMO(
const MachineBasicBlock::iterator &MI) const {
@@ -898,7 +916,7 @@ std::optional<SIMemOpInfo> SIMemOpAccess::constructFromMIWithMMO(
return SIMemOpInfo(ST, Ordering, Scope, OrderingAddrSpace, InstrAddrSpace,
IsCrossAddressSpaceOrdering, FailureOrdering, IsVolatile,
IsNonTemporal, IsLastUse, IsCooperative,
- hasAVNoneMMRA(*MI));
+ hasAVNoneMMRA(*MI), CanDemoteWorkgroupToWavefront);
}
std::optional<SIMemOpInfo>
@@ -968,7 +986,7 @@ SIMemOpAccess::getAtomicFenceInfo(const MachineBasicBlock::iterator &MI) const {
return SIMemOpInfo(ST, Ordering, Scope, OrderingAddrSpace,
SIAtomicAddrSpace::ATOMIC, IsCrossAddressSpaceOrdering,
AtomicOrdering::NotAtomic, false, false, false, false,
- hasAVNoneMMRA(*MI));
+ hasAVNoneMMRA(*MI), CanDemoteWorkgroupToWavefront);
}
std::optional<SIMemOpInfo> SIMemOpAccess::getAtomicCmpxchgOrRmwInfo(
@@ -2586,7 +2604,7 @@ bool SIMemoryLegalizer::run(MachineFunction &MF) {
const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
const Function &F = MF.getFunction();
- SIMemOpAccess MOA(MMI.getObjFileInfo<AMDGPUMachineModuleInfo>(), ST);
+ SIMemOpAccess MOA(MMI.getObjFileInfo<AMDGPUMachineModuleInfo>(), ST, F);
bool TgSplit = ST.hasTgSplitSupport() && AMDGPU::isTgSplitEnabled(F);
CC = SICacheControl::create(ST, TgSplit);
diff --git a/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll b/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll
index f40b209b77da7a..d83c8d133e7276 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll
@@ -6750,7 +6750,6 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_max_i32 v[0:1], v2
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_max_saddr_i32_nortn:
@@ -6764,7 +6763,6 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_i32 v[2:3], v1
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_max_saddr_i32_nortn:
@@ -6773,7 +6771,6 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: v_mov_b32_e32 v1, 0
; GFX950-SDAG-NEXT: v_lshl_add_u64 v[0:1], s[2:3], 0, v[0:1]
; GFX950-SDAG-NEXT: flat_atomic_smax v[0:1], v2
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_max_saddr_i32_nortn:
@@ -6783,7 +6780,6 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, 0, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_smax v[2:3], v1
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -6802,7 +6798,6 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_max_i32 v[0:1], v2 offset:-128
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_max_saddr_i32_nortn_neg128:
@@ -6816,7 +6811,6 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_i32 v[2:3], v1 offset:-128
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_max_saddr_i32_nortn_neg128:
@@ -6828,7 +6822,6 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_addc_co_u32_e32 v1, vcc, -1, v1, vcc
; GFX950-SDAG-NEXT: flat_atomic_smax v[0:1], v2
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_max_saddr_i32_nortn_neg128:
@@ -6841,7 +6834,6 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, -1, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_smax v[2:3], v1
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -6873,16 +6865,18 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB58_4
; GFX1250-SDAG-NEXT: .LBB58_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB58_5
; GFX1250-SDAG-NEXT: .LBB58_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_max_i64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB58_2
; GFX1250-SDAG-NEXT: .LBB58_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -6920,16 +6914,18 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB58_4
; GFX1250-GISEL-NEXT: .LBB58_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB58_5
; GFX1250-GISEL-NEXT: .LBB58_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_max_i64 v[0:1], v[2:3], v[4:5] th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB58_2
; GFX1250-GISEL-NEXT: .LBB58_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -6960,11 +6956,10 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB58_4
; GFX950-SDAG-NEXT: .LBB58_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB58_5
; GFX950-SDAG-NEXT: .LBB58_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smax_x2 v[0:1], v[4:5], v[2:3] sc0
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -6973,6 +6968,7 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_gt_i64_e32 vcc, v[0:1], v[2:3]
@@ -7004,11 +7000,10 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB58_4
; GFX950-GISEL-NEXT: .LBB58_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB58_5
; GFX950-GISEL-NEXT: .LBB58_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smax_x2 v[0:1], v[2:3], v[4:5] sc0
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7017,6 +7012,7 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_gt_i64_e32 vcc, v[0:1], v[4:5]
@@ -7061,16 +7057,18 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB59_4
; GFX1250-SDAG-NEXT: .LBB59_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB59_5
; GFX1250-SDAG-NEXT: .LBB59_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_max_i64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB59_2
; GFX1250-SDAG-NEXT: .LBB59_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -7111,16 +7109,18 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB59_4
; GFX1250-GISEL-NEXT: .LBB59_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB59_5
; GFX1250-GISEL-NEXT: .LBB59_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_max_i64 v[0:1], v[6:7], v[4:5] offset:-128 th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB59_2
; GFX1250-GISEL-NEXT: .LBB59_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -7154,11 +7154,10 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB59_4
; GFX950-SDAG-NEXT: .LBB59_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB59_5
; GFX950-SDAG-NEXT: .LBB59_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smax_x2 v[0:1], v[4:5], v[2:3] sc0
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7167,6 +7166,7 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_gt_i64_e32 vcc, v[0:1], v[2:3]
@@ -7201,11 +7201,10 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB59_4
; GFX950-GISEL-NEXT: .LBB59_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB59_5
; GFX950-GISEL-NEXT: .LBB59_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smax_x2 v[0:1], v[2:3], v[4:5] sc0
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7214,6 +7213,7 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_gt_i64_e32 vcc, v[0:1], v[4:5]
@@ -7259,7 +7259,6 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: flat_atomic_max_i64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB60_2
@@ -7301,7 +7300,6 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: flat_atomic_max_i64 v[0:1], v[4:5]
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB60_2
@@ -7335,7 +7333,6 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB60_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smax_x2 v[0:1], v[2:3]
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7372,7 +7369,6 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB60_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smax_x2 v[0:1], v[4:5]
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7423,7 +7419,6 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: flat_atomic_max_i64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB61_2
@@ -7468,7 +7463,6 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: flat_atomic_max_i64 v[2:3], v[4:5] offset:-128
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB61_2
@@ -7505,7 +7499,6 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB61_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smax_x2 v[0:1], v[2:3]
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7546,7 +7539,6 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB61_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smax_x2 v[0:1], v[4:5]
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7698,7 +7690,6 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_min_i32 v[0:1], v2
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_min_saddr_i32_nortn:
@@ -7712,7 +7703,6 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_i32 v[2:3], v1
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_min_saddr_i32_nortn:
@@ -7721,7 +7711,6 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: v_mov_b32_e32 v1, 0
; GFX950-SDAG-NEXT: v_lshl_add_u64 v[0:1], s[2:3], 0, v[0:1]
; GFX950-SDAG-NEXT: flat_atomic_smin v[0:1], v2
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_min_saddr_i32_nortn:
@@ -7731,7 +7720,6 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, 0, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_smin v[2:3], v1
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -7750,7 +7738,6 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_min_i32 v[0:1], v2 offset:-128
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_min_saddr_i32_nortn_neg128:
@@ -7764,7 +7751,6 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_i32 v[2:3], v1 offset:-128
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_min_saddr_i32_nortn_neg128:
@@ -7776,7 +7762,6 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_addc_co_u32_e32 v1, vcc, -1, v1, vcc
; GFX950-SDAG-NEXT: flat_atomic_smin v[0:1], v2
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_min_saddr_i32_nortn_neg128:
@@ -7789,7 +7774,6 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, -1, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_smin v[2:3], v1
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -7821,16 +7805,18 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB66_4
; GFX1250-SDAG-NEXT: .LBB66_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB66_5
; GFX1250-SDAG-NEXT: .LBB66_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_min_i64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB66_2
; GFX1250-SDAG-NEXT: .LBB66_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -7868,16 +7854,18 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB66_4
; GFX1250-GISEL-NEXT: .LBB66_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB66_5
; GFX1250-GISEL-NEXT: .LBB66_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_min_i64 v[0:1], v[2:3], v[4:5] th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB66_2
; GFX1250-GISEL-NEXT: .LBB66_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -7908,11 +7896,10 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB66_4
; GFX950-SDAG-NEXT: .LBB66_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB66_5
; GFX950-SDAG-NEXT: .LBB66_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smin_x2 v[0:1], v[4:5], v[2:3] sc0
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7921,6 +7908,7 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_le_i64_e32 vcc, v[0:1], v[2:3]
@@ -7952,11 +7940,10 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB66_4
; GFX950-GISEL-NEXT: .LBB66_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB66_5
; GFX950-GISEL-NEXT: .LBB66_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smin_x2 v[0:1], v[2:3], v[4:5] sc0
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7965,6 +7952,7 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_lt_i64_e32 vcc, v[0:1], v[4:5]
@@ -8009,16 +7997,18 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB67_4
; GFX1250-SDAG-NEXT: .LBB67_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB67_5
; GFX1250-SDAG-NEXT: .LBB67_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_min_i64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB67_2
; GFX1250-SDAG-NEXT: .LBB67_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -8059,16 +8049,18 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB67_4
; GFX1250-GISEL-NEXT: .LBB67_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB67_5
; GFX1250-GISEL-NEXT: .LBB67_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_min_i64 v[0:1], v[6:7], v[4:5] offset:-128 th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB67_2
; GFX1250-GISEL-NEXT: .LBB67_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -8102,11 +8094,10 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB67_4
; GFX950-SDAG-NEXT: .LBB67_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB67_5
; GFX950-SDAG-NEXT: .LBB67_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smin_x2 v[0:1], v[4:5], v[2:3] sc0
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8115,6 +8106,7 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_le_i64_e32 vcc, v[0:1], v[2:3]
@@ -8149,11 +8141,10 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB67_4
; GFX950-GISEL-NEXT: .LBB67_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB67_5
; GFX950-GISEL-NEXT: .LBB67_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smin_x2 v[0:1], v[2:3], v[4:5] sc0
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8162,6 +8153,7 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_lt_i64_e32 vcc, v[0:1], v[4:5]
@@ -8207,7 +8199,6 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: flat_atomic_min_i64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB68_2
@@ -8249,7 +8240,6 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: flat_atomic_min_i64 v[0:1], v[4:5]
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB68_2
@@ -8283,7 +8273,6 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB68_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smin_x2 v[0:1], v[2:3]
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8320,7 +8309,6 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB68_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smin_x2 v[0:1], v[4:5]
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8371,7 +8359,6 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: flat_atomic_min_i64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB69_2
@@ -8416,7 +8403,6 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: flat_atomic_min_i64 v[2:3], v[4:5] offset:-128
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB69_2
@@ -8453,7 +8439,6 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB69_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smin_x2 v[0:1], v[2:3]
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8494,7 +8479,6 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB69_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smin_x2 v[0:1], v[4:5]
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8646,7 +8630,6 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_max_u32 v[0:1], v2
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_umax_saddr_i32_nortn:
@@ -8660,7 +8643,6 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_u32 v[2:3], v1
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_umax_saddr_i32_nortn:
@@ -8669,7 +8651,6 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: v_mov_b32_e32 v1, 0
; GFX950-SDAG-NEXT: v_lshl_add_u64 v[0:1], s[2:3], 0, v[0:1]
; GFX950-SDAG-NEXT: flat_atomic_umax v[0:1], v2
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_umax_saddr_i32_nortn:
@@ -8679,7 +8660,6 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, 0, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_umax v[2:3], v1
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -8698,7 +8678,6 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_max_u32 v[0:1], v2 offset:-128
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_umax_saddr_i32_nortn_neg128:
@@ -8712,7 +8691,6 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_u32 v[2:3], v1 offset:-128
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_umax_saddr_i32_nortn_neg128:
@@ -8724,7 +8702,6 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_addc_co_u32_e32 v1, vcc, -1, v1, vcc
; GFX950-SDAG-NEXT: flat_atomic_umax v[0:1], v2
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_umax_saddr_i32_nortn_neg128:
@@ -8737,7 +8714,6 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, -1, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_umax v[2:3], v1
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -8769,16 +8745,18 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB74_4
; GFX1250-SDAG-NEXT: .LBB74_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB74_5
; GFX1250-SDAG-NEXT: .LBB74_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_max_u64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB74_2
; GFX1250-SDAG-NEXT: .LBB74_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -8816,16 +8794,18 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB74_4
; GFX1250-GISEL-NEXT: .LBB74_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB74_5
; GFX1250-GISEL-NEXT: .LBB74_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_max_u64 v[0:1], v[2:3], v[4:5] th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB74_2
; GFX1250-GISEL-NEXT: .LBB74_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -8856,11 +8836,10 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB74_4
; GFX950-SDAG-NEXT: .LBB74_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB74_5
; GFX950-SDAG-NEXT: .LBB74_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umax_x2 v[0:1], v[4:5], v[2:3] sc0
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8869,6 +8848,7 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_gt_u64_e32 vcc, v[0:1], v[2:3]
@@ -8900,11 +8880,10 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB74_4
; GFX950-GISEL-NEXT: .LBB74_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB74_5
; GFX950-GISEL-NEXT: .LBB74_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umax_x2 v[0:1], v[2:3], v[4:5] sc0
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8913,6 +8892,7 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_gt_u64_e32 vcc, v[0:1], v[4:5]
@@ -8957,16 +8937,18 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB75_4
; GFX1250-SDAG-NEXT: .LBB75_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB75_5
; GFX1250-SDAG-NEXT: .LBB75_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_max_u64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB75_2
; GFX1250-SDAG-NEXT: .LBB75_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -9007,16 +8989,18 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB75_4
; GFX1250-GISEL-NEXT: .LBB75_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB75_5
; GFX1250-GISEL-NEXT: .LBB75_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_max_u64 v[0:1], v[6:7], v[4:5] offset:-128 th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB75_2
; GFX1250-GISEL-NEXT: .LBB75_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -9050,11 +9034,10 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB75_4
; GFX950-SDAG-NEXT: .LBB75_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB75_5
; GFX950-SDAG-NEXT: .LBB75_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umax_x2 v[0:1], v[4:5], v[2:3] sc0
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9063,6 +9046,7 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_gt_u64_e32 vcc, v[0:1], v[2:3]
@@ -9097,11 +9081,10 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB75_4
; GFX950-GISEL-NEXT: .LBB75_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB75_5
; GFX950-GISEL-NEXT: .LBB75_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umax_x2 v[0:1], v[2:3], v[4:5] sc0
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9110,6 +9093,7 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_gt_u64_e32 vcc, v[0:1], v[4:5]
@@ -9155,7 +9139,6 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: flat_atomic_max_u64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB76_2
@@ -9197,7 +9180,6 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: flat_atomic_max_u64 v[0:1], v[4:5]
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB76_2
@@ -9231,7 +9213,6 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB76_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umax_x2 v[0:1], v[2:3]
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9268,7 +9249,6 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB76_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umax_x2 v[0:1], v[4:5]
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9319,7 +9299,6 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-SDAG-NEXT: flat_atomic_max_u64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB77_2
@@ -9364,7 +9343,6 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: flat_atomic_max_u64 v[2:3], v[4:5] offset:-128
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB77_2
@@ -9401,7 +9379,6 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB77_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umax_x2 v[0:1], v[2:3]
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9442,7 +9419,6 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB77_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umax_x2 v[0:1], v[4:5]
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9594,7 +9570,6 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_min_u32 v[0:1], v2
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_umin_saddr_i32_nortn:
@@ -9608,7 +9583,6 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_u32 v[2:3], v1
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_umin_saddr_i32_nortn:
@@ -9617,7 +9591,6 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: v_mov_b32_e32 v1, 0
; GFX950-SDAG-NEXT: v_lshl_add_u64 v[0:1], s[2:3], 0, v[0:1]
; GFX950-SDAG-NEXT: flat_atomic_umin v[0:1], v2
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_umin_saddr_i32_nortn:
@@ -9627,7 +9600,6 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, 0, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_umin v[2:3], v1
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -9646,7 +9618,6 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_min_u32 v[0:1], v2 offset:-128
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_umin_saddr_i32_nortn_neg128:
@@ -9660,7 +9631,6 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_u32 v[2:3], v1 offset:-128
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_umin_saddr_i32_nortn_neg128:
@@ -9672,7 +9642,6 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_addc_co_u32_e32 v1, vcc, -1, v1, vcc
; GFX950-SDAG-NEXT: flat_atomic_umin v[0:1], v2
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_umin_saddr_i32_nortn_neg128:
@@ -9685,7 +9654,6 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, -1, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_umin v[2:3], v1
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -9717,16 +9685,18 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB82_4
; GFX1250-SDAG-NEXT: .LBB82_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB82_5
; GFX1250-SDAG-NEXT: .LBB82_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_min_u64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB82_2
; GFX1250-SDAG-NEXT: .LBB82_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -9764,16 +9734,18 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB82_4
; GFX1250-GISEL-NEXT: .LBB82_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB82_5
; GFX1250-GISEL-NEXT: .LBB82_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_min_u64 v[0:1], v[2:3], v[4:5] th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB82_2
; GFX1250-GISEL-NEXT: .LBB82_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -9804,11 +9776,10 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB82_4
; GFX950-SDAG-NEXT: .LBB82_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB82_5
; GFX950-SDAG-NEXT: .LBB82_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umin_x2 v[0:1], v[4:5], v[2:3] sc0
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9817,6 +9788,7 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_le_u64_e32 vcc, v[0:1], v[2:3]
@@ -9848,11 +9820,10 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB82_4
; GFX950-GISEL-NEXT: .LBB82_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB82_5
; GFX950-GISEL-NEXT: .LBB82_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umin_x2 v[0:1], v[2:3], v[4:5] sc0
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9861,6 +9832,7 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_lt_u64_e32 vcc, v[0:1], v[4:5]
@@ -9905,16 +9877,18 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB83_4
; GFX1250-SDAG-NEXT: .LBB83_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB83_5
; GFX1250-SDAG-NEXT: .LBB83_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_min_u64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB83_2
; GFX1250-SDAG-NEXT: .LBB83_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -9955,16 +9929,18 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB83_4
; GFX1250-GISEL-NEXT: .LBB83_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB83_5
; GFX1250-GISEL-NEXT: .LBB83_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_min_u64 v[0:1], v[6:7], v[4:5] offset:-128 th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB83_2
; GFX1250-GISEL-NEXT: .LBB83_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -9998,11 +9974,10 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB83_4
; GFX950-SDAG-NEXT: .LBB83_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB83_5
; GFX950-SDAG-NEXT: .LBB83_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umin_x2 v[0:1], v[4:5], v[2:3] sc0
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -10011,6 +9986,7 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_le_u64_e32 vcc, v[0:1], v[2:3]
@@ -10045,11 +10021,10 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB83_4
; GFX950-GISEL-NEXT: .LBB83_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB83_5
; GFX950-GISEL-NEXT: .LBB83_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umin_x2 v[0:1], v[2:3], v[4:5] sc0
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -10058,6 +10033,7 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_lt_u64_e32 vcc, v[0:1], v[4:5]
@@ -10103,7 +10079,6 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: flat_atomic_min_u64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB84_2
@@ -10145,7 +10120,6 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: flat_atomic_min_u64 v[0:1], v[4:5]
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB84_2
@@ -10179,7 +10153,6 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB84_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umin_x2 v[0:1], v[2:3]
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -10216,7 +10189,6 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB84_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umin_x2 v[0:1], v[4:5]
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -10267,7 +10239,6 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-SDAG-NEXT: flat_atomic_min_u64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB85_2
@@ -10312,7 +10283,6 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: flat_atomic_min_u64 v[2:3], v[4:5] offset:-128
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB85_2
@@ -10349,7 +10319,6 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB85_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umin_x2 v[0:1], v[2:3]
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -10390,7 +10359,6 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB85_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umin_x2 v[0:1], v[4:5]
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/global-saddr-atomics.ll b/llvm/test/CodeGen/AMDGPU/global-saddr-atomics.ll
index c9ebcc15242580..e728825cc2f5ef 100644
--- a/llvm/test/CodeGen/AMDGPU/global-saddr-atomics.ll
+++ b/llvm/test/CodeGen/AMDGPU/global-saddr-atomics.ll
@@ -2142,31 +2142,22 @@ define amdgpu_ps void @global_xor_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; --------------------------------------------------------------------------------
define amdgpu_ps float @global_max_saddr_i32_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_max_saddr_i32_rtn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smax v0, v0, v1, s[2:3] glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_max_saddr_i32_rtn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smax v0, v0, v1, s[2:3] glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_max_saddr_i32_rtn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smax v0, v0, v1, s[2:3] glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_max_saddr_i32_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i32 v0, v0, v1, s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_max_saddr_i32_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_i32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2176,31 +2167,22 @@ define amdgpu_ps float @global_max_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps float @global_max_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_max_saddr_i32_rtn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smax v0, v0, v1, s[2:3] offset:-128 glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_max_saddr_i32_rtn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smax v0, v0, v1, s[2:3] offset:-128 glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_max_saddr_i32_rtn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smax v0, v0, v1, s[2:3] offset:-128 glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_max_saddr_i32_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i32 v0, v0, v1, s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_max_saddr_i32_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_i32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2211,30 +2193,19 @@ define amdgpu_ps float @global_max_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %
}
define amdgpu_ps void @global_max_saddr_i32_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_max_saddr_i32_nortn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smax v0, v1, s[2:3]
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_max_saddr_i32_nortn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smax v0, v1, s[2:3]
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_max_saddr_i32_nortn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smax v0, v1, s[2:3]
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_max_saddr_i32_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i32 v0, v1, s[2:3]
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_max_saddr_i32_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i32 v0, v1, s[2:3] scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_i32 v0, v1, s[2:3]
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2243,30 +2214,19 @@ define amdgpu_ps void @global_max_saddr_i32_nortn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps void @global_max_saddr_i32_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_max_saddr_i32_nortn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smax v0, v1, s[2:3] offset:-128
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_max_saddr_i32_nortn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smax v0, v1, s[2:3] offset:-128
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_max_saddr_i32_nortn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smax v0, v1, s[2:3] offset:-128
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_max_saddr_i32_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i32 v0, v1, s[2:3] offset:-128
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_max_saddr_i32_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i32 v0, v1, s[2:3] offset:-128 scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_i32 v0, v1, s[2:3] offset:-128
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2276,31 +2236,22 @@ define amdgpu_ps void @global_max_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
}
define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_max_saddr_i64_rtn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smax_x2 v[0:1], v0, v[1:2], s[2:3] glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_max_saddr_i64_rtn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smax_x2 v[0:1], v0, v[1:2], s[2:3] glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_max_saddr_i64_rtn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smax_x2 v[0:1], v0, v[1:2], s[2:3] glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_max_saddr_i64_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i64 v[0:1], v0, v[1:2], s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_max_saddr_i64_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_i64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2310,31 +2261,22 @@ define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn(ptr addrspace(1) inreg %s
}
define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_max_saddr_i64_rtn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smax_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_max_saddr_i64_rtn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smax_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_max_saddr_i64_rtn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smax_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_max_saddr_i64_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i64 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_max_saddr_i64_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_i64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2345,30 +2287,19 @@ define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn_neg128(ptr addrspace(1) i
}
define amdgpu_ps void @global_max_saddr_i64_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_max_saddr_i64_nortn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smax_x2 v0, v[1:2], s[2:3]
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_max_saddr_i64_nortn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smax_x2 v0, v[1:2], s[2:3]
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_max_saddr_i64_nortn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smax_x2 v0, v[1:2], s[2:3]
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_max_saddr_i64_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i64 v0, v[1:2], s[2:3]
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_max_saddr_i64_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i64 v0, v[1:2], s[2:3] scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_i64 v0, v[1:2], s[2:3]
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2377,30 +2308,19 @@ define amdgpu_ps void @global_max_saddr_i64_nortn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps void @global_max_saddr_i64_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_max_saddr_i64_nortn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smax_x2 v0, v[1:2], s[2:3] offset:-128
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_max_saddr_i64_nortn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smax_x2 v0, v[1:2], s[2:3] offset:-128
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_max_saddr_i64_nortn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smax_x2 v0, v[1:2], s[2:3] offset:-128
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_max_saddr_i64_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i64 v0, v[1:2], s[2:3] offset:-128
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_max_saddr_i64_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i64 v0, v[1:2], s[2:3] offset:-128 scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_i64 v0, v[1:2], s[2:3] offset:-128
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2414,31 +2334,22 @@ define amdgpu_ps void @global_max_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; --------------------------------------------------------------------------------
define amdgpu_ps float @global_min_saddr_i32_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_min_saddr_i32_rtn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smin v0, v0, v1, s[2:3] glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_min_saddr_i32_rtn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smin v0, v0, v1, s[2:3] glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_min_saddr_i32_rtn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smin v0, v0, v1, s[2:3] glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_min_saddr_i32_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i32 v0, v0, v1, s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_min_saddr_i32_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_i32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2448,31 +2359,22 @@ define amdgpu_ps float @global_min_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps float @global_min_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_min_saddr_i32_rtn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smin v0, v0, v1, s[2:3] offset:-128 glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_min_saddr_i32_rtn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smin v0, v0, v1, s[2:3] offset:-128 glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_min_saddr_i32_rtn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smin v0, v0, v1, s[2:3] offset:-128 glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_min_saddr_i32_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i32 v0, v0, v1, s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_min_saddr_i32_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_i32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2483,30 +2385,19 @@ define amdgpu_ps float @global_min_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %
}
define amdgpu_ps void @global_min_saddr_i32_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_min_saddr_i32_nortn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smin v0, v1, s[2:3]
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_min_saddr_i32_nortn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smin v0, v1, s[2:3]
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_min_saddr_i32_nortn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smin v0, v1, s[2:3]
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_min_saddr_i32_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i32 v0, v1, s[2:3]
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_min_saddr_i32_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i32 v0, v1, s[2:3] scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_i32 v0, v1, s[2:3]
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2515,30 +2406,19 @@ define amdgpu_ps void @global_min_saddr_i32_nortn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps void @global_min_saddr_i32_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_min_saddr_i32_nortn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smin v0, v1, s[2:3] offset:-128
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_min_saddr_i32_nortn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smin v0, v1, s[2:3] offset:-128
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_min_saddr_i32_nortn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smin v0, v1, s[2:3] offset:-128
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_min_saddr_i32_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i32 v0, v1, s[2:3] offset:-128
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_min_saddr_i32_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i32 v0, v1, s[2:3] offset:-128 scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_i32 v0, v1, s[2:3] offset:-128
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2548,31 +2428,22 @@ define amdgpu_ps void @global_min_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
}
define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_min_saddr_i64_rtn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smin_x2 v[0:1], v0, v[1:2], s[2:3] glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_min_saddr_i64_rtn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smin_x2 v[0:1], v0, v[1:2], s[2:3] glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_min_saddr_i64_rtn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smin_x2 v[0:1], v0, v[1:2], s[2:3] glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_min_saddr_i64_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i64 v[0:1], v0, v[1:2], s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_min_saddr_i64_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_i64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2582,31 +2453,22 @@ define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn(ptr addrspace(1) inreg %s
}
define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_min_saddr_i64_rtn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smin_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_min_saddr_i64_rtn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smin_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_min_saddr_i64_rtn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smin_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_min_saddr_i64_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i64 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_min_saddr_i64_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_i64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2617,30 +2479,19 @@ define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn_neg128(ptr addrspace(1) i
}
define amdgpu_ps void @global_min_saddr_i64_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_min_saddr_i64_nortn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smin_x2 v0, v[1:2], s[2:3]
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_min_saddr_i64_nortn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smin_x2 v0, v[1:2], s[2:3]
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_min_saddr_i64_nortn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smin_x2 v0, v[1:2], s[2:3]
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_min_saddr_i64_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i64 v0, v[1:2], s[2:3]
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_min_saddr_i64_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i64 v0, v[1:2], s[2:3] scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_i64 v0, v[1:2], s[2:3]
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2649,30 +2500,19 @@ define amdgpu_ps void @global_min_saddr_i64_nortn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps void @global_min_saddr_i64_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_min_saddr_i64_nortn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_smin_x2 v0, v[1:2], s[2:3] offset:-128
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_min_saddr_i64_nortn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_smin_x2 v0, v[1:2], s[2:3] offset:-128
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_min_saddr_i64_nortn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_smin_x2 v0, v[1:2], s[2:3] offset:-128
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_min_saddr_i64_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i64 v0, v[1:2], s[2:3] offset:-128
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_min_saddr_i64_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i64 v0, v[1:2], s[2:3] offset:-128 scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_i64 v0, v[1:2], s[2:3] offset:-128
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2686,31 +2526,22 @@ define amdgpu_ps void @global_min_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; --------------------------------------------------------------------------------
define amdgpu_ps float @global_umax_saddr_i32_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_umax_saddr_i32_rtn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umax v0, v0, v1, s[2:3] glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_umax_saddr_i32_rtn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umax v0, v0, v1, s[2:3] glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_umax_saddr_i32_rtn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umax v0, v0, v1, s[2:3] glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umax_saddr_i32_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u32 v0, v0, v1, s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umax_saddr_i32_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_u32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2720,31 +2551,22 @@ define amdgpu_ps float @global_umax_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps float @global_umax_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_umax_saddr_i32_rtn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umax v0, v0, v1, s[2:3] offset:-128 glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_umax_saddr_i32_rtn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umax v0, v0, v1, s[2:3] offset:-128 glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_umax_saddr_i32_rtn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umax v0, v0, v1, s[2:3] offset:-128 glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umax_saddr_i32_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u32 v0, v0, v1, s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umax_saddr_i32_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_u32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2755,30 +2577,19 @@ define amdgpu_ps float @global_umax_saddr_i32_rtn_neg128(ptr addrspace(1) inreg
}
define amdgpu_ps void @global_umax_saddr_i32_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_umax_saddr_i32_nortn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umax v0, v1, s[2:3]
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_umax_saddr_i32_nortn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umax v0, v1, s[2:3]
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_umax_saddr_i32_nortn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umax v0, v1, s[2:3]
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_umax_saddr_i32_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u32 v0, v1, s[2:3]
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umax_saddr_i32_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u32 v0, v1, s[2:3] scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_u32 v0, v1, s[2:3]
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2787,30 +2598,19 @@ define amdgpu_ps void @global_umax_saddr_i32_nortn(ptr addrspace(1) inreg %sbase
}
define amdgpu_ps void @global_umax_saddr_i32_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_umax_saddr_i32_nortn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umax v0, v1, s[2:3] offset:-128
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_umax_saddr_i32_nortn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umax v0, v1, s[2:3] offset:-128
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_umax_saddr_i32_nortn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umax v0, v1, s[2:3] offset:-128
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_umax_saddr_i32_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u32 v0, v1, s[2:3] offset:-128
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umax_saddr_i32_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u32 v0, v1, s[2:3] offset:-128 scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_u32 v0, v1, s[2:3] offset:-128
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2820,31 +2620,22 @@ define amdgpu_ps void @global_umax_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
}
define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_umax_saddr_i64_rtn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umax_x2 v[0:1], v0, v[1:2], s[2:3] glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_umax_saddr_i64_rtn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umax_x2 v[0:1], v0, v[1:2], s[2:3] glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_umax_saddr_i64_rtn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umax_x2 v[0:1], v0, v[1:2], s[2:3] glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umax_saddr_i64_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u64 v[0:1], v0, v[1:2], s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umax_saddr_i64_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_u64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2854,31 +2645,22 @@ define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn(ptr addrspace(1) inreg %
}
define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_umax_saddr_i64_rtn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umax_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_umax_saddr_i64_rtn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umax_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_umax_saddr_i64_rtn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umax_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umax_saddr_i64_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u64 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umax_saddr_i64_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_u64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2889,30 +2671,19 @@ define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn_neg128(ptr addrspace(1)
}
define amdgpu_ps void @global_umax_saddr_i64_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_umax_saddr_i64_nortn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umax_x2 v0, v[1:2], s[2:3]
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_umax_saddr_i64_nortn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umax_x2 v0, v[1:2], s[2:3]
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_umax_saddr_i64_nortn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umax_x2 v0, v[1:2], s[2:3]
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_umax_saddr_i64_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u64 v0, v[1:2], s[2:3]
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umax_saddr_i64_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u64 v0, v[1:2], s[2:3] scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_u64 v0, v[1:2], s[2:3]
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2921,30 +2692,19 @@ define amdgpu_ps void @global_umax_saddr_i64_nortn(ptr addrspace(1) inreg %sbase
}
define amdgpu_ps void @global_umax_saddr_i64_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_umax_saddr_i64_nortn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umax_x2 v0, v[1:2], s[2:3] offset:-128
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_umax_saddr_i64_nortn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umax_x2 v0, v[1:2], s[2:3] offset:-128
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_umax_saddr_i64_nortn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umax_x2 v0, v[1:2], s[2:3] offset:-128
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_umax_saddr_i64_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u64 v0, v[1:2], s[2:3] offset:-128
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umax_saddr_i64_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u64 v0, v[1:2], s[2:3] offset:-128 scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_max_u64 v0, v[1:2], s[2:3] offset:-128
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2958,31 +2718,22 @@ define amdgpu_ps void @global_umax_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; --------------------------------------------------------------------------------
define amdgpu_ps float @global_umin_saddr_i32_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_umin_saddr_i32_rtn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umin v0, v0, v1, s[2:3] glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_umin_saddr_i32_rtn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umin v0, v0, v1, s[2:3] glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_umin_saddr_i32_rtn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umin v0, v0, v1, s[2:3] glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umin_saddr_i32_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u32 v0, v0, v1, s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umin_saddr_i32_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_u32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2992,31 +2743,22 @@ define amdgpu_ps float @global_umin_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps float @global_umin_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_umin_saddr_i32_rtn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umin v0, v0, v1, s[2:3] offset:-128 glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_umin_saddr_i32_rtn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umin v0, v0, v1, s[2:3] offset:-128 glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_umin_saddr_i32_rtn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umin v0, v0, v1, s[2:3] offset:-128 glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umin_saddr_i32_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u32 v0, v0, v1, s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umin_saddr_i32_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_u32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -3027,30 +2769,19 @@ define amdgpu_ps float @global_umin_saddr_i32_rtn_neg128(ptr addrspace(1) inreg
}
define amdgpu_ps void @global_umin_saddr_i32_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_umin_saddr_i32_nortn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umin v0, v1, s[2:3]
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_umin_saddr_i32_nortn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umin v0, v1, s[2:3]
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_umin_saddr_i32_nortn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umin v0, v1, s[2:3]
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_umin_saddr_i32_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u32 v0, v1, s[2:3]
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umin_saddr_i32_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u32 v0, v1, s[2:3] scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_u32 v0, v1, s[2:3]
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -3059,30 +2790,19 @@ define amdgpu_ps void @global_umin_saddr_i32_nortn(ptr addrspace(1) inreg %sbase
}
define amdgpu_ps void @global_umin_saddr_i32_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GFX9-LABEL: global_umin_saddr_i32_nortn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umin v0, v1, s[2:3] offset:-128
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_umin_saddr_i32_nortn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umin v0, v1, s[2:3] offset:-128
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_umin_saddr_i32_nortn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umin v0, v1, s[2:3] offset:-128
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_umin_saddr_i32_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u32 v0, v1, s[2:3] offset:-128
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umin_saddr_i32_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u32 v0, v1, s[2:3] offset:-128 scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_u32 v0, v1, s[2:3] offset:-128
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -3092,31 +2812,22 @@ define amdgpu_ps void @global_umin_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
}
define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_umin_saddr_i64_rtn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umin_x2 v[0:1], v0, v[1:2], s[2:3] glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_umin_saddr_i64_rtn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umin_x2 v[0:1], v0, v[1:2], s[2:3] glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_umin_saddr_i64_rtn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umin_x2 v[0:1], v0, v[1:2], s[2:3] glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umin_saddr_i64_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u64 v[0:1], v0, v[1:2], s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umin_saddr_i64_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_u64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -3126,31 +2837,22 @@ define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn(ptr addrspace(1) inreg %
}
define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_umin_saddr_i64_rtn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umin_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
-; GFX9-NEXT: s_waitcnt vmcnt(0)
-; GFX9-NEXT: ; return to shader part epilog
-;
-; GFX10-LABEL: global_umin_saddr_i64_rtn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umin_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
-; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: ; return to shader part epilog
+; GCN-LABEL: global_umin_saddr_i64_rtn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umin_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umin_saddr_i64_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u64 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umin_saddr_i64_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_u64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -3161,30 +2863,19 @@ define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn_neg128(ptr addrspace(1)
}
define amdgpu_ps void @global_umin_saddr_i64_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_umin_saddr_i64_nortn:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umin_x2 v0, v[1:2], s[2:3]
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_umin_saddr_i64_nortn:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umin_x2 v0, v[1:2], s[2:3]
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_umin_saddr_i64_nortn:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umin_x2 v0, v[1:2], s[2:3]
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_umin_saddr_i64_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u64 v0, v[1:2], s[2:3]
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umin_saddr_i64_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u64 v0, v[1:2], s[2:3] scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_u64 v0, v[1:2], s[2:3]
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -3193,30 +2884,19 @@ define amdgpu_ps void @global_umin_saddr_i64_nortn(ptr addrspace(1) inreg %sbase
}
define amdgpu_ps void @global_umin_saddr_i64_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GFX9-LABEL: global_umin_saddr_i64_nortn_neg128:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: global_atomic_umin_x2 v0, v[1:2], s[2:3] offset:-128
-; GFX9-NEXT: s_endpgm
-;
-; GFX10-LABEL: global_umin_saddr_i64_nortn_neg128:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: global_atomic_umin_x2 v0, v[1:2], s[2:3] offset:-128
-; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX10-NEXT: buffer_gl0_inv
-; GFX10-NEXT: s_endpgm
+; GCN-LABEL: global_umin_saddr_i64_nortn_neg128:
+; GCN: ; %bb.0:
+; GCN-NEXT: global_atomic_umin_x2 v0, v[1:2], s[2:3] offset:-128
+; GCN-NEXT: s_endpgm
;
; GFX11-LABEL: global_umin_saddr_i64_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u64 v0, v[1:2], s[2:3] offset:-128
-; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
-; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umin_saddr_i64_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u64 v0, v[1:2], s[2:3] offset:-128 scope:SCOPE_SE
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_inv scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_min_u64 v0, v[1:2], s[2:3] offset:-128
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
diff --git a/llvm/test/CodeGen/AMDGPU/lds-dma-war-async.ll b/llvm/test/CodeGen/AMDGPU/lds-dma-war-async.ll
index c89f41f8100608..8be725c7bef2cc 100644
--- a/llvm/test/CodeGen/AMDGPU/lds-dma-war-async.ll
+++ b/llvm/test/CodeGen/AMDGPU/lds-dma-war-async.ll
@@ -45,8 +45,8 @@ define i32 @lds_then_dma.global.wg(ptr addrspace(1) %g, ptr addrspace(3) inreg %
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: v_mov_b32_e32 v3, s0
; CHECK-NEXT: ds_load_b32 v2, v3
-; CHECK-NEXT: s_wait_dscnt 0x0
; CHECK-NEXT: global_load_async_to_lds_b32 v3, v[0:1], off
+; CHECK-NEXT: s_wait_dscnt 0x0
; CHECK-NEXT: v_mov_b32_e32 v0, v2
; CHECK-NEXT: s_set_pc_i64 s[30:31]
%v = load i32, ptr addrspace(3) %lds, align 4
@@ -136,7 +136,6 @@ define i32 @lds_then_dma.global.wg.partial(ptr addrspace(1) %g, ptr addrspace(3)
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v4, s1
; CHECK-NEXT: ds_load_b32 v3, v2
-; CHECK-NEXT: s_wait_dscnt 0x0
; CHECK-NEXT: ds_load_b32 v4, v4
; CHECK-NEXT: global_load_async_to_lds_b32 v2, v[0:1], off
; CHECK-NEXT: s_wait_dscnt 0x0
@@ -231,8 +230,8 @@ define i32 @lds_then_dma.tensor.wg(<4 x i32> inreg %D0, <8 x i32> inreg %D1, ptr
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: v_mov_b32_e32 v0, s24
; CHECK-NEXT: ds_load_b32 v0, v0
-; CHECK-NEXT: s_wait_dscnt 0x0
; CHECK-NEXT: tensor_load_to_lds s[0:3], s[16:23]
+; CHECK-NEXT: s_wait_dscnt 0x0
; CHECK-NEXT: s_set_pc_i64 s[30:31]
%v = load i32, ptr addrspace(3) %lds, align 4
fence syncscope("workgroup") release, !mmra !0
@@ -319,7 +318,6 @@ define i32 @lds_then_dma.tensor.wg.partial(<4 x i32> inreg %D0, <8 x i32> inreg
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: v_dual_mov_b32 v0, s24 :: v_dual_mov_b32 v1, s25
; CHECK-NEXT: ds_load_b32 v0, v0
-; CHECK-NEXT: s_wait_dscnt 0x0
; CHECK-NEXT: ds_load_b32 v1, v1
; CHECK-NEXT: tensor_load_to_lds s[0:3], s[16:23]
; CHECK-NEXT: s_wait_dscnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/lds-dma-war.ll b/llvm/test/CodeGen/AMDGPU/lds-dma-war.ll
index 77b44f304abfed..6ee1b2ed83b808 100644
--- a/llvm/test/CodeGen/AMDGPU/lds-dma-war.ll
+++ b/llvm/test/CodeGen/AMDGPU/lds-dma-war.ll
@@ -47,8 +47,8 @@ define i32 @lds_then_dma.global.wg(ptr addrspace(1) %g, ptr addrspace(3) inreg %
; CHECK-NEXT: v_mov_b32_e32 v2, s0
; CHECK-NEXT: s_mov_b32 m0, s0
; CHECK-NEXT: ds_read_b32 v2, v2
-; CHECK-NEXT: s_waitcnt lgkmcnt(0)
; CHECK-NEXT: global_load_lds_dword v[0:1], off
+; CHECK-NEXT: s_waitcnt lgkmcnt(0)
; CHECK-NEXT: v_mov_b32_e32 v0, v2
; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: s_setpc_b64 s[30:31]
@@ -146,7 +146,6 @@ define i32 @lds_then_dma.global.wg.partial(ptr addrspace(1) %g, ptr addrspace(3)
; CHECK-NEXT: v_mov_b32_e32 v2, s0
; CHECK-NEXT: v_mov_b32_e32 v3, s1
; CHECK-NEXT: ds_read_b32 v2, v2
-; CHECK-NEXT: s_waitcnt lgkmcnt(0)
; CHECK-NEXT: ds_read_b32 v3, v3
; CHECK-NEXT: global_load_lds_dword v[0:1], off
; CHECK-NEXT: s_waitcnt lgkmcnt(0)
@@ -246,9 +245,8 @@ define i32 @lds_then_dma.buffer.wg(<4 x i32> inreg %rsrc, ptr addrspace(3) inreg
; CHECK-NEXT: v_mov_b32_e32 v0, s16
; CHECK-NEXT: s_mov_b32 m0, s16
; CHECK-NEXT: ds_read_b32 v0, v0
-; CHECK-NEXT: s_waitcnt lgkmcnt(0)
; CHECK-NEXT: buffer_load_dword off, s[0:3], 0 lds
-; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; CHECK-NEXT: s_setpc_b64 s[30:31]
%v = load i32, ptr addrspace(3) %lds, align 4
fence syncscope("workgroup") release, !mmra !0
@@ -342,7 +340,6 @@ define i32 @lds_then_dma.buffer.wg.partial(<4 x i32> inreg %rsrc, ptr addrspace(
; CHECK-NEXT: v_mov_b32_e32 v0, s16
; CHECK-NEXT: v_mov_b32_e32 v1, s17
; CHECK-NEXT: ds_read_b32 v0, v0
-; CHECK-NEXT: s_waitcnt lgkmcnt(0)
; CHECK-NEXT: ds_read_b32 v1, v1
; CHECK-NEXT: buffer_load_dword off, s[0:3], 0 lds
; CHECK-NEXT: s_waitcnt lgkmcnt(0)
diff --git a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma-async.ll b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma-async.ll
index 13efb8daae0a3a..826436c2ca5bc8 100644
--- a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma-async.ll
+++ b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma-async.ll
@@ -13,9 +13,6 @@ define amdgpu_kernel void @lds_wg_fence_release_single32(ptr addrspace(3) %lds)
; GFX1250-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 32 :: (dereferenceable invariant load (s32) from %ir.lds.kernarg.offset, addrspace 4)
; GFX1250-NEXT: renamable $vgpr0, $vgpr1 = V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1250 1, killed $sgpr0, implicit $exec, implicit $exec, implicit $exec, implicit $exec
; GFX1250-NEXT: DS_WRITE_B32_gfx9 renamable $vgpr1, renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
- ; GFX1250-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX1250-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX1250-NEXT: S_WAIT_DSCNT_soft 0
; GFX1250-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
; GFX1250-NEXT: S_ENDPGM 0
store i32 1, ptr addrspace(3) %lds, align 4
@@ -35,9 +32,6 @@ define amdgpu_kernel void @lds_async_dma_wg_fence_release_single32(ptr addrspace
; GFX1250-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2 = S_LOAD_DWORDX3_IMM_ec killed renamable $sgpr4_sgpr5, 36, 32 :: (dereferenceable invariant load (s96) from %ir.g.kernarg.offset, align 4, addrspace 4)
; GFX1250-NEXT: renamable $vgpr0, $vgpr1 = V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1250 0, killed $sgpr2, implicit $exec, implicit $exec, implicit $exec, implicit $exec
; GFX1250-NEXT: GLOBAL_LOAD_ASYNC_TO_LDS_B32_SADDR killed $vgpr1, killed $sgpr0_sgpr1, killed $vgpr0, 0, 0, implicit-def dead $asynccnt, implicit $exec, implicit $asynccnt :: (load (s32) from %ir.g.load, align 1, addrspace 1), (store (s32) into %ir.lds.load, align 1, addrspace 3)
- ; GFX1250-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX1250-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX1250-NEXT: S_WAIT_DSCNT_soft 0
; GFX1250-NEXT: S_ENDPGM 0
call void @llvm.amdgcn.global.load.async.to.lds.b32(ptr addrspace(1) %g, ptr addrspace(3) %lds, i32 0, i32 0)
fence syncscope("workgroup") release
diff --git a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma.ll b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma.ll
index 2cfe2bb0ad0bd1..5944db56e6a3b3 100644
--- a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma.ll
+++ b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma.ll
@@ -17,8 +17,6 @@ define amdgpu_kernel void @lds_wg_fence_release_single32(ptr addrspace(3) %lds)
; GFX9-NEXT: renamable $vgpr0 = V_MOV_B32_e32 1, implicit $exec
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX9-NEXT: DS_WRITE_B32_gfx9 renamable $vgpr1, renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
; GFX9-NEXT: S_ENDPGM 0
;
@@ -32,8 +30,6 @@ define amdgpu_kernel void @lds_wg_fence_release_single32(ptr addrspace(3) %lds)
; GFX942-NEXT: renamable $vgpr0 = V_MOV_B32_e32 1, implicit $exec
; GFX942-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX942-NEXT: DS_WRITE_B32_gfx9 renamable $vgpr1, renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
; GFX942-NEXT: S_ENDPGM 0
;
@@ -47,9 +43,6 @@ define amdgpu_kernel void @lds_wg_fence_release_single32(ptr addrspace(3) %lds)
; GFX10-NEXT: renamable $vgpr0 = V_MOV_B32_e32 1, implicit $exec
; GFX10-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX10-NEXT: DS_WRITE_B32_gfx9 renamable $vgpr1, renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
; GFX10-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
; GFX10-NEXT: S_ENDPGM 0
store i32 1, ptr addrspace(3) %lds, align 4
@@ -70,8 +63,6 @@ define amdgpu_kernel void @lds_wg_fence_release_single64(ptr addrspace(3) %lds)
; GFX9-NEXT: renamable $vgpr0 = V_MOV_B32_e32 1, implicit $exec
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX9-NEXT: DS_WRITE_B32_gfx9 renamable $vgpr1, renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
; GFX9-NEXT: S_ENDPGM 0
;
@@ -85,26 +76,37 @@ define amdgpu_kernel void @lds_wg_fence_release_single64(ptr addrspace(3) %lds)
; GFX942-NEXT: renamable $vgpr0 = V_MOV_B32_e32 1, implicit $exec
; GFX942-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX942-NEXT: DS_WRITE_B32_gfx9 renamable $vgpr1, renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: lds_wg_fence_release_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.lds.kernarg.offset, addrspace 4)
- ; GFX10-NEXT: renamable $vgpr0 = V_MOV_B32_e32 1, implicit $exec
- ; GFX10-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX10-NEXT: DS_WRITE_B32_gfx9 renamable $vgpr1, renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: lds_wg_fence_release_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.lds.kernarg.offset, addrspace 4)
+ ; GFX10-W32-NEXT: renamable $vgpr0 = V_MOV_B32_e32 1, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: DS_WRITE_B32_gfx9 renamable $vgpr1, renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
+ ; GFX10-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX10-W64-LABEL: name: lds_wg_fence_release_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.lds.kernarg.offset, addrspace 4)
+ ; GFX10-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 1, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: DS_WRITE_B32_gfx9 renamable $vgpr1, renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
+ ; GFX10-W64-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: (store (s32) into %ir.lds.load, addrspace 3)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
store i32 1, ptr addrspace(3) %lds, align 4
fence syncscope("workgroup") release
%v = load i32, ptr addrspace(3) %lds, align 4
@@ -123,8 +125,6 @@ define amdgpu_kernel void @lds_dma_wg_fence_release_single32(ptr addrspace(8) %r
; GFX9-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.rsrc.kernarg.offset, align 4, addrspace 4)
; GFX9-NEXT: $m0 = S_MOV_B32 killed $sgpr6
; GFX9-NEXT: BUFFER_LOAD_DWORD_LDS_OFFSET killed $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, 0, implicit $exec, implicit $m0 :: (dereferenceable load (s32) from %ir.rsrc.load, align 1, addrspace 8), (dereferenceable store (s2048) into %ir.lds.load, align 1, addrspace 3)
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: S_ENDPGM 0
;
; GFX942-LABEL: name: lds_dma_wg_fence_release_single32
@@ -137,8 +137,6 @@ define amdgpu_kernel void @lds_dma_wg_fence_release_single32(ptr addrspace(8) %r
; GFX942-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.rsrc.kernarg.offset, align 4, addrspace 4)
; GFX942-NEXT: $m0 = S_MOV_B32 killed $sgpr6
; GFX942-NEXT: BUFFER_LOAD_DWORD_LDS_OFFSET killed $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, 0, implicit $exec, implicit $m0 :: (dereferenceable load (s32) from %ir.rsrc.load, align 1, addrspace 8), (dereferenceable store (s2048) into %ir.lds.load, align 1, addrspace 3)
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: S_ENDPGM 0
;
; GFX10-W32-LABEL: name: lds_dma_wg_fence_release_single32
@@ -151,9 +149,6 @@ define amdgpu_kernel void @lds_dma_wg_fence_release_single32(ptr addrspace(8) %r
; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.rsrc.kernarg.offset, align 4, addrspace 4)
; GFX10-W32-NEXT: $m0 = S_MOV_B32 killed $sgpr6
; GFX10-W32-NEXT: BUFFER_LOAD_DWORD_LDS_OFFSET killed $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, 0, implicit $exec, implicit $m0 :: (dereferenceable load (s32) from %ir.rsrc.load, align 1, addrspace 8), (dereferenceable store (s1024) into %ir.lds.load, align 1, addrspace 3)
- ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
- ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
; GFX10-W32-NEXT: S_ENDPGM 0
;
; GFX10-W64-LABEL: name: lds_dma_wg_fence_release_single32
@@ -166,9 +161,6 @@ define amdgpu_kernel void @lds_dma_wg_fence_release_single32(ptr addrspace(8) %r
; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.rsrc.kernarg.offset, align 4, addrspace 4)
; GFX10-W64-NEXT: $m0 = S_MOV_B32 killed $sgpr6
; GFX10-W64-NEXT: BUFFER_LOAD_DWORD_LDS_OFFSET killed $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, 0, implicit $exec, implicit $m0 :: (dereferenceable load (s32) from %ir.rsrc.load, align 1, addrspace 8), (dereferenceable store (s2048) into %ir.lds.load, align 1, addrspace 3)
- ; GFX10-W64-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-W64-NEXT: S_WAITCNT_lds_direct
- ; GFX10-W64-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
; GFX10-W64-NEXT: S_ENDPGM 0
call void @llvm.amdgcn.raw.ptr.buffer.load.lds(ptr addrspace(8) %rsrc, ptr addrspace(3) %lds, i32 4, i32 0, i32 0, i32 0, i32 0)
fence syncscope("workgroup") release
diff --git a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-memops.ll b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-memops.ll
index ab89739b75bd47..a4c0d1f535b864 100644
--- a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-memops.ll
+++ b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-memops.ll
@@ -13,47 +13,30 @@ define amdgpu_kernel void @wg_fence_acq_rel_single32() #0 {
; GFX9: bb.0 (%ir-block.0):
; GFX9-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX9-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: S_ENDPGM 0
;
; GFX942-LABEL: name: wg_fence_acq_rel_single32
; GFX942: bb.0 (%ir-block.0):
; GFX942-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX942-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: S_ENDPGM 0
;
; GFX10-LABEL: name: wg_fence_acq_rel_single32
; GFX10: bb.0 (%ir-block.0):
; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
; GFX10-NEXT: S_ENDPGM 0
;
; GFX12-LABEL: name: wg_fence_acq_rel_single32
; GFX12: bb.0 (%ir-block.0):
; GFX12-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX12-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX12-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-NEXT: GLOBAL_INV 8, implicit $exec
; GFX12-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_fence_acq_rel_single32
; GFX1250: bb.0 (%ir-block.0):
; GFX1250-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX1250-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX1250-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX1250-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX1250-NEXT: S_WAIT_DSCNT_soft 0
; GFX1250-NEXT: S_ENDPGM 0
fence syncscope("workgroup") acq_rel
ret void
@@ -64,39 +47,47 @@ define amdgpu_kernel void @wg_fence_acq_rel_single64() #1 {
; GFX9: bb.0 (%ir-block.0):
; GFX9-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX9-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: S_ENDPGM 0
;
; GFX942-LABEL: name: wg_fence_acq_rel_single64
; GFX942: bb.0 (%ir-block.0):
; GFX942-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX942-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: wg_fence_acq_rel_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: wg_fence_acq_rel_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-W32-NEXT: S_ENDPGM 0
;
- ; GFX12-LABEL: name: wg_fence_acq_rel_single64
- ; GFX12: bb.0 (%ir-block.0):
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX12-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-NEXT: GLOBAL_INV 8, implicit $exec
- ; GFX12-NEXT: S_ENDPGM 0
+ ; GFX10-W64-LABEL: name: wg_fence_acq_rel_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W32-LABEL: name: wg_fence_acq_rel_single64
+ ; GFX12-W32: bb.0 (%ir-block.0):
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W32-NEXT: S_WAIT_BVHCNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_SAMPLECNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_LOADCNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_STORECNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_DSCNT_soft 0
+ ; GFX12-W32-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W64-LABEL: name: wg_fence_acq_rel_single64
+ ; GFX12-W64: bb.0 (%ir-block.0):
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_fence_acq_rel_single64
; GFX1250: bb.0 (%ir-block.0):
@@ -166,34 +157,44 @@ define amdgpu_kernel void @wg_fence_acquire_single64() #1 {
; GFX9: bb.0 (%ir-block.0):
; GFX9-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX9-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
; GFX9-NEXT: S_ENDPGM 0
;
; GFX942-LABEL: name: wg_fence_acquire_single64
; GFX942: bb.0 (%ir-block.0):
; GFX942-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX942-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: wg_fence_acquire_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: wg_fence_acquire_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-W32-NEXT: S_ENDPGM 0
;
- ; GFX12-LABEL: name: wg_fence_acquire_single64
- ; GFX12: bb.0 (%ir-block.0):
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX12-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-NEXT: GLOBAL_INV 8, implicit $exec
- ; GFX12-NEXT: S_ENDPGM 0
+ ; GFX10-W64-LABEL: name: wg_fence_acquire_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W32-LABEL: name: wg_fence_acquire_single64
+ ; GFX12-W32: bb.0 (%ir-block.0):
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W32-NEXT: S_WAIT_LOADCNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_STORECNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_DSCNT_soft 0
+ ; GFX12-W32-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W64-LABEL: name: wg_fence_acquire_single64
+ ; GFX12-W64: bb.0 (%ir-block.0):
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_fence_acquire_single64
; GFX1250: bb.0 (%ir-block.0):
@@ -212,37 +213,45 @@ define amdgpu_kernel void @wg_fence_release_single64() #1 {
; GFX9: bb.0 (%ir-block.0):
; GFX9-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX9-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: S_ENDPGM 0
;
; GFX942-LABEL: name: wg_fence_release_single64
; GFX942: bb.0 (%ir-block.0):
; GFX942-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX942-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: wg_fence_release_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: wg_fence_release_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: S_ENDPGM 0
;
- ; GFX12-LABEL: name: wg_fence_release_single64
- ; GFX12: bb.0 (%ir-block.0):
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX12-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-NEXT: S_ENDPGM 0
+ ; GFX10-W64-LABEL: name: wg_fence_release_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W32-LABEL: name: wg_fence_release_single64
+ ; GFX12-W32: bb.0 (%ir-block.0):
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W32-NEXT: S_WAIT_BVHCNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_SAMPLECNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_LOADCNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_STORECNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_DSCNT_soft 0
+ ; GFX12-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W64-LABEL: name: wg_fence_release_single64
+ ; GFX12-W64: bb.0 (%ir-block.0):
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_fence_release_single64
; GFX1250: bb.0 (%ir-block.0):
@@ -261,39 +270,47 @@ define amdgpu_kernel void @wg_fence_seq_cst_single64() #1 {
; GFX9: bb.0 (%ir-block.0):
; GFX9-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX9-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: S_ENDPGM 0
;
; GFX942-LABEL: name: wg_fence_seq_cst_single64
; GFX942: bb.0 (%ir-block.0):
; GFX942-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
; GFX942-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: wg_fence_seq_cst_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: wg_fence_seq_cst_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-W32-NEXT: S_ENDPGM 0
;
- ; GFX12-LABEL: name: wg_fence_seq_cst_single64
- ; GFX12: bb.0 (%ir-block.0):
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX12-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-NEXT: GLOBAL_INV 8, implicit $exec
- ; GFX12-NEXT: S_ENDPGM 0
+ ; GFX10-W64-LABEL: name: wg_fence_seq_cst_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W32-LABEL: name: wg_fence_seq_cst_single64
+ ; GFX12-W32: bb.0 (%ir-block.0):
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W32-NEXT: S_WAIT_BVHCNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_SAMPLECNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_LOADCNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_STORECNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_DSCNT_soft 0
+ ; GFX12-W32-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W64-LABEL: name: wg_fence_seq_cst_single64
+ ; GFX12-W64: bb.0 (%ir-block.0):
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_fence_seq_cst_single64
; GFX1250: bb.0 (%ir-block.0):
@@ -316,8 +333,6 @@ define amdgpu_kernel void @wg_ld_seq_cst_single32(ptr addrspace(1) %p) #0 {
; GFX9-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX9-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX9-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
; GFX9-NEXT: S_ENDPGM 0
;
@@ -329,9 +344,7 @@ define amdgpu_kernel void @wg_ld_seq_cst_single32(ptr addrspace(1) %p) #0 {
; GFX942-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX942-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX942-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
- ; GFX942-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 1, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
+ ; GFX942-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
; GFX942-NEXT: S_ENDPGM 0
;
; GFX10-LABEL: name: wg_ld_seq_cst_single32
@@ -342,12 +355,7 @@ define amdgpu_kernel void @wg_ld_seq_cst_single32(ptr addrspace(1) %p) #0 {
; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX10-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 1, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
; GFX10-NEXT: S_ENDPGM 0
;
; GFX12-LABEL: name: wg_ld_seq_cst_single32
@@ -358,14 +366,7 @@ define amdgpu_kernel void @wg_ld_seq_cst_single32(ptr addrspace(1) %p) #0 {
; GFX12-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX12-NEXT: renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX12-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX12-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 8, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
- ; GFX12-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
; GFX12-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_ld_seq_cst_single32
@@ -376,11 +377,7 @@ define amdgpu_kernel void @wg_ld_seq_cst_single32(ptr addrspace(1) %p) #0 {
; GFX1250-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX1250-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 32 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX1250-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX1250-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX1250-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX1250-NEXT: S_WAIT_DSCNT_soft 0
; GFX1250-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
- ; GFX1250-NEXT: S_WAIT_LOADCNT_soft 0
; GFX1250-NEXT: S_ENDPGM 0
%v = load atomic i32, ptr addrspace(1) %p syncscope("workgroup") seq_cst, align 4
ret void
@@ -395,8 +392,6 @@ define amdgpu_kernel void @wg_ld_seq_cst_single64(ptr addrspace(1) %p) #1 {
; GFX9-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX9-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX9-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
; GFX9-NEXT: S_ENDPGM 0
;
@@ -408,44 +403,64 @@ define amdgpu_kernel void @wg_ld_seq_cst_single64(ptr addrspace(1) %p) #1 {
; GFX942-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX942-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX942-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
- ; GFX942-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 1, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
+ ; GFX942-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: wg_ld_seq_cst_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX10-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 1, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: wg_ld_seq_cst_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W32-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 1, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0
+ ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-W32-NEXT: S_ENDPGM 0
;
- ; GFX12-LABEL: name: wg_ld_seq_cst_single64
- ; GFX12: bb.0 (%ir-block.0):
- ; GFX12-NEXT: liveins: $sgpr4_sgpr5
- ; GFX12-NEXT: {{ $}}
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX12-NEXT: renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX12-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX12-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 8, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
- ; GFX12-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-NEXT: GLOBAL_INV 8, implicit $exec
- ; GFX12-NEXT: S_ENDPGM 0
+ ; GFX10-W64-LABEL: name: wg_ld_seq_cst_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W64-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W32-LABEL: name: wg_ld_seq_cst_single64
+ ; GFX12-W32: bb.0 (%ir-block.0):
+ ; GFX12-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX12-W32-NEXT: {{ $}}
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W32-NEXT: renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX12-W32-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX12-W32-NEXT: S_WAIT_BVHCNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_SAMPLECNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_LOADCNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_STORECNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_DSCNT_soft 0
+ ; GFX12-W32-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 8, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
+ ; GFX12-W32-NEXT: S_WAIT_LOADCNT_soft 0
+ ; GFX12-W32-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W64-LABEL: name: wg_ld_seq_cst_single64
+ ; GFX12-W64: bb.0 (%ir-block.0):
+ ; GFX12-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX12-W64-NEXT: {{ $}}
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W64-NEXT: renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX12-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX12-W64-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 1)
+ ; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_ld_seq_cst_single64
; GFX1250: bb.0 (%ir-block.0):
@@ -564,34 +579,56 @@ define amdgpu_kernel void @wg_ld_acquire_single64(ptr addrspace(1) %p) #1 {
; GFX942-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX942-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX942-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX942-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 1, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") acquire (s32) from %ir.p.load, addrspace 1)
+ ; GFX942-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") acquire (s32) from %ir.p.load, addrspace 1)
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: wg_ld_acquire_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX10-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX10-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 1, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") acquire (s32) from %ir.p.load, addrspace 1)
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: wg_ld_acquire_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W32-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W32-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 1, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") acquire (s32) from %ir.p.load, addrspace 1)
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0
+ ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-W32-NEXT: S_ENDPGM 0
;
- ; GFX12-LABEL: name: wg_ld_acquire_single64
- ; GFX12: bb.0 (%ir-block.0):
- ; GFX12-NEXT: liveins: $sgpr4_sgpr5
- ; GFX12-NEXT: {{ $}}
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX12-NEXT: renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX12-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX12-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 8, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") acquire (s32) from %ir.p.load, addrspace 1)
- ; GFX12-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-NEXT: GLOBAL_INV 8, implicit $exec
- ; GFX12-NEXT: S_ENDPGM 0
+ ; GFX10-W64-LABEL: name: wg_ld_acquire_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W64-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") acquire (s32) from %ir.p.load, addrspace 1)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W32-LABEL: name: wg_ld_acquire_single64
+ ; GFX12-W32: bb.0 (%ir-block.0):
+ ; GFX12-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX12-W32-NEXT: {{ $}}
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W32-NEXT: renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX12-W32-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX12-W32-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 8, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") acquire (s32) from %ir.p.load, addrspace 1)
+ ; GFX12-W32-NEXT: S_WAIT_LOADCNT_soft 0
+ ; GFX12-W32-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W64-LABEL: name: wg_ld_acquire_single64
+ ; GFX12-W64: bb.0 (%ir-block.0):
+ ; GFX12-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX12-W64-NEXT: {{ $}}
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W64-NEXT: renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX12-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX12-W64-NEXT: dead renamable $vgpr0 = GLOBAL_LOAD_DWORD_SADDR killed renamable $sgpr0_sgpr1, killed renamable $vgpr0, 0, 0, implicit $exec :: ("amdgpu-noclobber" load syncscope("workgroup") acquire (s32) from %ir.p.load, addrspace 1)
+ ; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_ld_acquire_single64
; GFX1250: bb.0 (%ir-block.0):
@@ -678,8 +715,6 @@ define amdgpu_kernel void @wg_st_seq_cst_single32(ptr addrspace(1) %p, i32 %x) #
; GFX9-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX9-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
; GFX9-NEXT: S_ENDPGM 0
;
@@ -693,9 +728,7 @@ define amdgpu_kernel void @wg_st_seq_cst_single32(ptr addrspace(1) %p, i32 %x) #
; GFX942-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX942-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX942-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
- ; GFX942-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 1, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
+ ; GFX942-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
; GFX942-NEXT: S_ENDPGM 0
;
; GFX10-LABEL: name: wg_st_seq_cst_single32
@@ -708,9 +741,6 @@ define amdgpu_kernel void @wg_st_seq_cst_single32(ptr addrspace(1) %p, i32 %x) #
; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX10-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX10-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
; GFX10-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
; GFX10-NEXT: S_ENDPGM 0
;
@@ -722,12 +752,7 @@ define amdgpu_kernel void @wg_st_seq_cst_single32(ptr addrspace(1) %p, i32 %x) #
; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX12-W32-NEXT: renamable $sgpr0_sgpr1_sgpr2 = S_LOAD_DWORDX3_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s96) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX12-W32-NEXT: renamable $vgpr0, $vgpr1 = V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx12 0, killed $sgpr2, implicit $exec, implicit $exec, implicit $exec, implicit $exec
- ; GFX12-W32-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-W32-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-W32-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W32-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W32-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W32-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 8, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
+ ; GFX12-W32-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
; GFX12-W32-NEXT: S_ENDPGM 0
;
; GFX12-W64-LABEL: name: wg_st_seq_cst_single32
@@ -739,12 +764,7 @@ define amdgpu_kernel void @wg_st_seq_cst_single32(ptr addrspace(1) %p, i32 %x) #
; GFX12-W64-NEXT: renamable $sgpr0_sgpr1_sgpr2 = S_LOAD_DWORDX3_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s96) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX12-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX12-W64-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 8, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
+ ; GFX12-W64-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_st_seq_cst_single32
@@ -755,9 +775,6 @@ define amdgpu_kernel void @wg_st_seq_cst_single32(ptr addrspace(1) %p, i32 %x) #
; GFX1250-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX1250-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2 = S_LOAD_DWORDX3_IMM_ec killed renamable $sgpr4_sgpr5, 36, 32 :: (dereferenceable invariant load (s96) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX1250-NEXT: renamable $vgpr0, $vgpr1 = V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1250 0, killed $sgpr2, implicit $exec, implicit $exec, implicit $exec, implicit $exec
- ; GFX1250-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX1250-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX1250-NEXT: S_WAIT_DSCNT_soft 0
; GFX1250-NEXT: S_WAIT_XCNT_soft 0
; GFX1250-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
; GFX1250-NEXT: S_ENDPGM 0
@@ -776,8 +793,6 @@ define amdgpu_kernel void @wg_st_seq_cst_single64(ptr addrspace(1) %p, i32 %x) #
; GFX9-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX9-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
; GFX9-NEXT: S_ENDPGM 0
;
@@ -791,26 +806,37 @@ define amdgpu_kernel void @wg_st_seq_cst_single64(ptr addrspace(1) %p, i32 %x) #
; GFX942-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX942-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX942-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
- ; GFX942-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 1, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
+ ; GFX942-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: wg_st_seq_cst_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: renamable $sgpr2 = S_LOAD_DWORD_IMM renamable $sgpr4_sgpr5, 44, 0 :: (dereferenceable invariant load (s32) from %ir.x.kernarg.offset, addrspace 4)
- ; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX10-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX10-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: wg_st_seq_cst_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: renamable $sgpr2 = S_LOAD_DWORD_IMM renamable $sgpr4_sgpr5, 44, 0 :: (dereferenceable invariant load (s32) from %ir.x.kernarg.offset, addrspace 4)
+ ; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W32-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
+ ; GFX10-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX10-W64-LABEL: name: wg_st_seq_cst_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: renamable $sgpr2 = S_LOAD_DWORD_IMM renamable $sgpr4_sgpr5, 44, 0 :: (dereferenceable invariant load (s32) from %ir.x.kernarg.offset, addrspace 4)
+ ; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
;
; GFX12-W32-LABEL: name: wg_st_seq_cst_single64
; GFX12-W32: bb.0 (%ir-block.0):
@@ -837,12 +863,7 @@ define amdgpu_kernel void @wg_st_seq_cst_single64(ptr addrspace(1) %p, i32 %x) #
; GFX12-W64-NEXT: renamable $sgpr0_sgpr1_sgpr2 = S_LOAD_DWORDX3_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s96) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX12-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX12-W64-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 8, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
+ ; GFX12-W64-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") seq_cst (s32) into %ir.p.load, addrspace 1)
; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_st_seq_cst_single64
@@ -972,8 +993,6 @@ define amdgpu_kernel void @wg_st_release_single64(ptr addrspace(1) %p, i32 %x) #
; GFX9-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX9-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.p.load, addrspace 1)
; GFX9-NEXT: S_ENDPGM 0
;
@@ -987,26 +1006,37 @@ define amdgpu_kernel void @wg_st_release_single64(ptr addrspace(1) %p, i32 %x) #
; GFX942-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX942-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX942-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
- ; GFX942-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 1, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.p.load, addrspace 1)
+ ; GFX942-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.p.load, addrspace 1)
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: wg_st_release_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: renamable $sgpr2 = S_LOAD_DWORD_IMM renamable $sgpr4_sgpr5, 44, 0 :: (dereferenceable invariant load (s32) from %ir.x.kernarg.offset, addrspace 4)
- ; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX10-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX10-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.p.load, addrspace 1)
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: wg_st_release_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: renamable $sgpr2 = S_LOAD_DWORD_IMM renamable $sgpr4_sgpr5, 44, 0 :: (dereferenceable invariant load (s32) from %ir.x.kernarg.offset, addrspace 4)
+ ; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W32-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.p.load, addrspace 1)
+ ; GFX10-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX10-W64-LABEL: name: wg_st_release_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: renamable $sgpr2 = S_LOAD_DWORD_IMM renamable $sgpr4_sgpr5, 44, 0 :: (dereferenceable invariant load (s32) from %ir.x.kernarg.offset, addrspace 4)
+ ; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.p.load, addrspace 1)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
;
; GFX12-W32-LABEL: name: wg_st_release_single64
; GFX12-W32: bb.0 (%ir-block.0):
@@ -1033,12 +1063,7 @@ define amdgpu_kernel void @wg_st_release_single64(ptr addrspace(1) %p, i32 %x) #
; GFX12-W64-NEXT: renamable $sgpr0_sgpr1_sgpr2 = S_LOAD_DWORDX3_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s96) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX12-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX12-W64-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 8, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.p.load, addrspace 1)
+ ; GFX12-W64-NEXT: GLOBAL_STORE_DWORD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.p.load, addrspace 1)
; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_st_release_single64
@@ -1083,8 +1108,6 @@ define amdgpu_kernel void @wg_rmw_add_seq_cst_single32(ptr addrspace(1) %p) #0 {
; GFX9-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 7
; GFX9-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 0, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
; GFX9-NEXT: {{ $}}
; GFX9-NEXT: bb.2 (%ir-block.16):
@@ -1113,8 +1136,6 @@ define amdgpu_kernel void @wg_rmw_add_seq_cst_single32(ptr addrspace(1) %p) #0 {
; GFX942-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 7
; GFX942-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX942-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 0, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
; GFX942-NEXT: {{ $}}
; GFX942-NEXT: bb.2 (%ir-block.16):
@@ -1142,12 +1163,7 @@ define amdgpu_kernel void @wg_rmw_add_seq_cst_single32(ptr addrspace(1) %p) #0 {
; GFX10-W32-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX10-W32-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 7
; GFX10-W32-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
- ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
; GFX10-W32-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 0, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
- ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
; GFX10-W32-NEXT: {{ $}}
; GFX10-W32-NEXT: bb.2 (%ir-block.11):
; GFX10-W32-NEXT: S_ENDPGM 0
@@ -1175,12 +1191,7 @@ define amdgpu_kernel void @wg_rmw_add_seq_cst_single32(ptr addrspace(1) %p) #0 {
; GFX10-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX10-W64-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 7
; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX10-W64-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-W64-NEXT: S_WAITCNT_lds_direct
- ; GFX10-W64-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
; GFX10-W64-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 0, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
- ; GFX10-W64-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-W64-NEXT: BUFFER_GL0_INV implicit $exec
; GFX10-W64-NEXT: {{ $}}
; GFX10-W64-NEXT: bb.2 (%ir-block.16):
; GFX10-W64-NEXT: S_ENDPGM 0
@@ -1206,14 +1217,7 @@ define amdgpu_kernel void @wg_rmw_add_seq_cst_single32(ptr addrspace(1) %p) #0 {
; GFX12-W32-NEXT: renamable $sgpr0 = S_BCNT1_I32_B32 killed renamable $sgpr0, implicit-def dead $scc
; GFX12-W32-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 7
; GFX12-W32-NEXT: renamable $vgpr0, $vgpr1 = V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx12 0, killed $sgpr0, implicit $exec, implicit $exec, implicit $exec, implicit $exec
- ; GFX12-W32-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-W32-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-W32-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W32-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W32-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W32-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 8, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
- ; GFX12-W32-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W32-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W32-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 0, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
; GFX12-W32-NEXT: {{ $}}
; GFX12-W32-NEXT: bb.2 (%ir-block.11):
; GFX12-W32-NEXT: S_ENDPGM 0
@@ -1241,14 +1245,7 @@ define amdgpu_kernel void @wg_rmw_add_seq_cst_single32(ptr addrspace(1) %p) #0 {
; GFX12-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX12-W64-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 7
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX12-W64-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 8, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W64-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 0, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
; GFX12-W64-NEXT: {{ $}}
; GFX12-W64-NEXT: bb.2 (%ir-block.16):
; GFX12-W64-NEXT: S_ENDPGM 0
@@ -1274,12 +1271,8 @@ define amdgpu_kernel void @wg_rmw_add_seq_cst_single32(ptr addrspace(1) %p) #0 {
; GFX1250-NEXT: renamable $sgpr0 = S_BCNT1_I32_B32 killed renamable $sgpr0, implicit-def dead $scc
; GFX1250-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 7
; GFX1250-NEXT: renamable $vgpr0, $vgpr1 = V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1250 0, killed $sgpr0, implicit $exec, implicit $exec, implicit $exec, implicit $exec
- ; GFX1250-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX1250-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX1250-NEXT: S_WAIT_DSCNT_soft 0
; GFX1250-NEXT: S_WAIT_XCNT_soft 0
; GFX1250-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 0, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
- ; GFX1250-NEXT: S_WAIT_STORECNT_soft 0
; GFX1250-NEXT: {{ $}}
; GFX1250-NEXT: bb.2 (%ir-block.11):
; GFX1250-NEXT: S_ENDPGM 0
@@ -1311,8 +1304,6 @@ define amdgpu_kernel void @wg_rmw_add_seq_cst_single64(ptr addrspace(1) %p) #1 {
; GFX9-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 7
; GFX9-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 0, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
; GFX9-NEXT: {{ $}}
; GFX9-NEXT: bb.2 (%ir-block.16):
@@ -1341,8 +1332,6 @@ define amdgpu_kernel void @wg_rmw_add_seq_cst_single64(ptr addrspace(1) %p) #1 {
; GFX942-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 7
; GFX942-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX942-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 0, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
; GFX942-NEXT: {{ $}}
; GFX942-NEXT: bb.2 (%ir-block.16):
@@ -1403,12 +1392,7 @@ define amdgpu_kernel void @wg_rmw_add_seq_cst_single64(ptr addrspace(1) %p) #1 {
; GFX10-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX10-W64-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 7
; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX10-W64-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-W64-NEXT: S_WAITCNT_lds_direct
- ; GFX10-W64-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
; GFX10-W64-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 0, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
- ; GFX10-W64-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-W64-NEXT: BUFFER_GL0_INV implicit $exec
; GFX10-W64-NEXT: {{ $}}
; GFX10-W64-NEXT: bb.2 (%ir-block.16):
; GFX10-W64-NEXT: S_ENDPGM 0
@@ -1469,14 +1453,7 @@ define amdgpu_kernel void @wg_rmw_add_seq_cst_single64(ptr addrspace(1) %p) #1 {
; GFX12-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX12-W64-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 7
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX12-W64-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 8, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W64-NEXT: GLOBAL_ATOMIC_ADD_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr2_sgpr3, 0, 0, implicit $exec :: (load store syncscope("workgroup") seq_cst (s32) on %ir.p.load, addrspace 1)
; GFX12-W64-NEXT: {{ $}}
; GFX12-W64-NEXT: bb.2 (%ir-block.16):
; GFX12-W64-NEXT: S_ENDPGM 0
@@ -1754,8 +1731,6 @@ define amdgpu_kernel void @wg_rmw_xchg_acq_rel_single64(ptr addrspace(1) %p, i32
; GFX9-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX9-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: GLOBAL_ATOMIC_SWAP_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel (s32) on %ir.p.load, addrspace 1)
; GFX9-NEXT: S_ENDPGM 0
;
@@ -1769,28 +1744,39 @@ define amdgpu_kernel void @wg_rmw_xchg_acq_rel_single64(ptr addrspace(1) %p, i32
; GFX942-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX942-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX942-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: GLOBAL_ATOMIC_SWAP_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel (s32) on %ir.p.load, addrspace 1)
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: wg_rmw_xchg_acq_rel_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: renamable $sgpr2 = S_LOAD_DWORD_IMM renamable $sgpr4_sgpr5, 44, 0 :: (dereferenceable invariant load (s32) from %ir.x.kernarg.offset, addrspace 4)
- ; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX10-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
- ; GFX10-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: GLOBAL_ATOMIC_SWAP_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel (s32) on %ir.p.load, addrspace 1)
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: wg_rmw_xchg_acq_rel_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: renamable $sgpr2 = S_LOAD_DWORD_IMM renamable $sgpr4_sgpr5, 44, 0 :: (dereferenceable invariant load (s32) from %ir.x.kernarg.offset, addrspace 4)
+ ; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W32-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: GLOBAL_ATOMIC_SWAP_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel (s32) on %ir.p.load, addrspace 1)
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX10-W64-LABEL: name: wg_rmw_xchg_acq_rel_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: renamable $sgpr2 = S_LOAD_DWORD_IMM renamable $sgpr4_sgpr5, 44, 0 :: (dereferenceable invariant load (s32) from %ir.x.kernarg.offset, addrspace 4)
+ ; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: GLOBAL_ATOMIC_SWAP_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel (s32) on %ir.p.load, addrspace 1)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
;
; GFX12-W32-LABEL: name: wg_rmw_xchg_acq_rel_single64
; GFX12-W32: bb.0 (%ir-block.0):
@@ -1819,14 +1805,7 @@ define amdgpu_kernel void @wg_rmw_xchg_acq_rel_single64(ptr addrspace(1) %p, i32
; GFX12-W64-NEXT: renamable $sgpr0_sgpr1_sgpr2 = S_LOAD_DWORDX3_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s96) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX12-W64-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX12-W64-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_ATOMIC_SWAP_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 8, implicit $exec :: (load store syncscope("workgroup") acq_rel (s32) on %ir.p.load, addrspace 1)
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W64-NEXT: GLOBAL_ATOMIC_SWAP_SADDR killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel (s32) on %ir.p.load, addrspace 1)
; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_rmw_xchg_acq_rel_single64
@@ -1859,8 +1838,6 @@ define amdgpu_kernel void @wg_cmpxchg_acq_rel_monotonic_single64(ptr addrspace(1
; GFX9-NEXT: renamable $vgpr2 = V_MOV_B32_e32 0, implicit $exec
; GFX9-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr3, implicit $exec, implicit $exec
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr2, killed renamable $vgpr0_vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.p.load, addrspace 1)
; GFX9-NEXT: S_ENDPGM 0
;
@@ -1874,28 +1851,39 @@ define amdgpu_kernel void @wg_cmpxchg_acq_rel_monotonic_single64(ptr addrspace(1
; GFX942-NEXT: renamable $vgpr0 = V_MOV_B32_e32 0, implicit $exec
; GFX942-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr3, implicit $exec, implicit $exec
; GFX942-NEXT: $vgpr3 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr0, killed renamable $vgpr2_vgpr3, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.p.load, addrspace 1)
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: wg_cmpxchg_acq_rel_monotonic_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX10-NEXT: renamable $vgpr2 = V_MOV_B32_e32 0, implicit $exec
- ; GFX10-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr3, implicit $exec, implicit $exec
- ; GFX10-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr2, killed renamable $vgpr0_vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.p.load, addrspace 1)
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: wg_cmpxchg_acq_rel_monotonic_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W32-NEXT: renamable $vgpr2 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr3, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr2, killed renamable $vgpr0_vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.p.load, addrspace 1)
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX10-W64-LABEL: name: wg_cmpxchg_acq_rel_monotonic_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W64-NEXT: renamable $vgpr2 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr3, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr2, killed renamable $vgpr0_vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.p.load, addrspace 1)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
;
; GFX12-W32-LABEL: name: wg_cmpxchg_acq_rel_monotonic_single64
; GFX12-W32: bb.0 (%ir-block.0):
@@ -1926,14 +1914,7 @@ define amdgpu_kernel void @wg_cmpxchg_acq_rel_monotonic_single64(ptr addrspace(1
; GFX12-W64-NEXT: renamable $vgpr2 = V_MOV_B32_e32 0, implicit $exec
; GFX12-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr3, implicit $exec, implicit $exec
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX12-W64-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr2, killed renamable $vgpr0_vgpr1, killed renamable $sgpr0_sgpr1, 0, 8, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.p.load, addrspace 1)
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W64-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr2, killed renamable $vgpr0_vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.p.load, addrspace 1)
; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_cmpxchg_acq_rel_monotonic_single64
@@ -2091,20 +2072,33 @@ define amdgpu_kernel void @wg_cmpxchg_acquire_acquire_single64(ptr addrspace(1)
; GFX942-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr0, killed renamable $vgpr2_vgpr3, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acquire acquire (s32) on %ir.p.load, addrspace 1)
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: wg_cmpxchg_acquire_acquire_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX10-NEXT: renamable $vgpr2 = V_MOV_B32_e32 0, implicit $exec
- ; GFX10-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr3, implicit $exec, implicit $exec
- ; GFX10-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX10-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr2, killed renamable $vgpr0_vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acquire acquire (s32) on %ir.p.load, addrspace 1)
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: wg_cmpxchg_acquire_acquire_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W32-NEXT: renamable $vgpr2 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr3, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr2, killed renamable $vgpr0_vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acquire acquire (s32) on %ir.p.load, addrspace 1)
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX10-W64-LABEL: name: wg_cmpxchg_acquire_acquire_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W64-NEXT: renamable $vgpr2 = V_MOV_B32_e32 0, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr3, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr2, killed renamable $vgpr0_vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acquire acquire (s32) on %ir.p.load, addrspace 1)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
;
; GFX12-W32-LABEL: name: wg_cmpxchg_acquire_acquire_single64
; GFX12-W32: bb.0 (%ir-block.0):
@@ -2130,9 +2124,7 @@ define amdgpu_kernel void @wg_cmpxchg_acquire_acquire_single64(ptr addrspace(1)
; GFX12-W64-NEXT: renamable $vgpr2 = V_MOV_B32_e32 0, implicit $exec
; GFX12-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr3, implicit $exec, implicit $exec
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX12-W64-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr2, killed renamable $vgpr0_vgpr1, killed renamable $sgpr0_sgpr1, 0, 8, implicit $exec :: (load store syncscope("workgroup") acquire acquire (s32) on %ir.p.load, addrspace 1)
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W64-NEXT: GLOBAL_ATOMIC_CMPSWAP_SADDR killed renamable $vgpr2, killed renamable $vgpr0_vgpr1, killed renamable $sgpr0_sgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acquire acquire (s32) on %ir.p.load, addrspace 1)
; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: wg_cmpxchg_acquire_acquire_single64
@@ -2161,11 +2153,7 @@ define amdgpu_kernel void @lds_wg_ld_seq_cst_single32(ptr addrspace(3) %p) #0 {
; GFX9-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX9-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
; GFX9-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: S_ENDPGM 0
;
; GFX942-LABEL: name: lds_wg_ld_seq_cst_single32
@@ -2176,11 +2164,7 @@ define amdgpu_kernel void @lds_wg_ld_seq_cst_single32(ptr addrspace(3) %p) #0 {
; GFX942-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX942-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
; GFX942-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: S_ENDPGM 0
;
; GFX10-LABEL: name: lds_wg_ld_seq_cst_single32
@@ -2191,13 +2175,7 @@ define amdgpu_kernel void @lds_wg_ld_seq_cst_single32(ptr addrspace(3) %p) #0 {
; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX10-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
; GFX10-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
; GFX10-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
- ; GFX10-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
; GFX10-NEXT: S_ENDPGM 0
;
; GFX12-LABEL: name: lds_wg_ld_seq_cst_single32
@@ -2208,14 +2186,7 @@ define amdgpu_kernel void @lds_wg_ld_seq_cst_single32(ptr addrspace(3) %p) #0 {
; GFX12-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX12-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
; GFX12-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX12-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-NEXT: S_WAIT_DSCNT_soft 0
; GFX12-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
- ; GFX12-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-NEXT: GLOBAL_INV 8, implicit $exec
; GFX12-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: lds_wg_ld_seq_cst_single32
@@ -2226,11 +2197,7 @@ define amdgpu_kernel void @lds_wg_ld_seq_cst_single32(ptr addrspace(3) %p) #0 {
; GFX1250-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX1250-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 32 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
; GFX1250-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX1250-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX1250-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX1250-NEXT: S_WAIT_DSCNT_soft 0
; GFX1250-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
- ; GFX1250-NEXT: S_WAIT_DSCNT_soft 0
; GFX1250-NEXT: S_ENDPGM 0
%v = load atomic i32, ptr addrspace(3) %p syncscope("workgroup") seq_cst, align 4
ret void
@@ -2245,11 +2212,7 @@ define amdgpu_kernel void @lds_wg_ld_seq_cst_single64(ptr addrspace(3) %p) #1 {
; GFX9-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX9-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
; GFX9-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: S_ENDPGM 0
;
; GFX942-LABEL: name: lds_wg_ld_seq_cst_single64
@@ -2260,47 +2223,65 @@ define amdgpu_kernel void @lds_wg_ld_seq_cst_single64(ptr addrspace(3) %p) #1 {
; GFX942-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX942-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
; GFX942-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: lds_wg_ld_seq_cst_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
- ; GFX10-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
- ; GFX10-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: lds_wg_ld_seq_cst_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
+ ; GFX10-W32-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-W32-NEXT: S_ENDPGM 0
;
- ; GFX12-LABEL: name: lds_wg_ld_seq_cst_single64
- ; GFX12: bb.0 (%ir-block.0):
- ; GFX12-NEXT: liveins: $sgpr4_sgpr5
- ; GFX12-NEXT: {{ $}}
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX12-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX12-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
- ; GFX12-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX12-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
- ; GFX12-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-NEXT: GLOBAL_INV 8, implicit $exec
- ; GFX12-NEXT: S_ENDPGM 0
+ ; GFX10-W64-LABEL: name: lds_wg_ld_seq_cst_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
+ ; GFX10-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W32-LABEL: name: lds_wg_ld_seq_cst_single64
+ ; GFX12-W32: bb.0 (%ir-block.0):
+ ; GFX12-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX12-W32-NEXT: {{ $}}
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W32-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
+ ; GFX12-W32-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
+ ; GFX12-W32-NEXT: S_WAIT_BVHCNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_SAMPLECNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_LOADCNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_STORECNT_soft 0
+ ; GFX12-W32-NEXT: S_WAIT_DSCNT_soft 0
+ ; GFX12-W32-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
+ ; GFX12-W32-NEXT: S_WAIT_DSCNT_soft 0
+ ; GFX12-W32-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX12-W64-LABEL: name: lds_wg_ld_seq_cst_single64
+ ; GFX12-W64: bb.0 (%ir-block.0):
+ ; GFX12-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX12-W64-NEXT: {{ $}}
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX12-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX12-W64-NEXT: renamable $sgpr0 = S_LOAD_DWORD_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s32) from %ir.p.kernarg.offset, addrspace 4)
+ ; GFX12-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
+ ; GFX12-W64-NEXT: dead renamable $vgpr0 = DS_READ_B32_gfx9 killed renamable $vgpr0, 0, 0, implicit $exec :: (load syncscope("workgroup") seq_cst (s32) from %ir.p.load, addrspace 3)
+ ; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: lds_wg_ld_seq_cst_single64
; GFX1250: bb.0 (%ir-block.0):
@@ -2414,8 +2395,6 @@ define amdgpu_kernel void @lds_wg_st_release_single64(ptr addrspace(3) %p, i32 %
; GFX9-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX9-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.2, addrspace 3)
; GFX9-NEXT: S_ENDPGM 0
;
@@ -2428,25 +2407,35 @@ define amdgpu_kernel void @lds_wg_st_release_single64(ptr addrspace(3) %p, i32 %
; GFX942-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX942-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX942-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.2, addrspace 3)
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: lds_wg_st_release_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX10-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX10-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.2, addrspace 3)
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: lds_wg_st_release_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W32-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.2, addrspace 3)
+ ; GFX10-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX10-W64-LABEL: name: lds_wg_st_release_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.2, addrspace 3)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
;
; GFX12-W32-LABEL: name: lds_wg_st_release_single64
; GFX12-W32: bb.0 (%ir-block.0):
@@ -2473,11 +2462,6 @@ define amdgpu_kernel void @lds_wg_st_release_single64(ptr addrspace(3) %p, i32 %
; GFX12-W64-NEXT: renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX12-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
- ; GFX12-W64-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
; GFX12-W64-NEXT: DS_WRITE_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, 0, 0, implicit $exec :: (store syncscope("workgroup") release (s32) into %ir.2, addrspace 3)
; GFX12-W64-NEXT: S_ENDPGM 0
;
@@ -2522,11 +2506,7 @@ define amdgpu_kernel void @lds_wg_rmw_add_acq_rel_single64(ptr addrspace(3) %p)
; GFX9-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 3
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX9-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: DS_ADD_U32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel (s32) on %ir.p.load, addrspace 3)
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: {{ $}}
; GFX9-NEXT: bb.2 (%ir-block.16):
; GFX9-NEXT: S_ENDPGM 0
@@ -2554,11 +2534,7 @@ define amdgpu_kernel void @lds_wg_rmw_add_acq_rel_single64(ptr addrspace(3) %p)
; GFX942-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 3
; GFX942-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX942-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: DS_ADD_U32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel (s32) on %ir.p.load, addrspace 3)
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: {{ $}}
; GFX942-NEXT: bb.2 (%ir-block.16):
; GFX942-NEXT: S_ENDPGM 0
@@ -2619,13 +2595,7 @@ define amdgpu_kernel void @lds_wg_rmw_add_acq_rel_single64(ptr addrspace(3) %p)
; GFX10-W64-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 3
; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX10-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX10-W64-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-W64-NEXT: S_WAITCNT_lds_direct
- ; GFX10-W64-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
; GFX10-W64-NEXT: DS_ADD_U32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel (s32) on %ir.p.load, addrspace 3)
- ; GFX10-W64-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX10-W64-NEXT: S_WAITCNT_lds_direct
- ; GFX10-W64-NEXT: BUFFER_GL0_INV implicit $exec
; GFX10-W64-NEXT: {{ $}}
; GFX10-W64-NEXT: bb.2 (%ir-block.16):
; GFX10-W64-NEXT: S_ENDPGM 0
@@ -2686,14 +2656,7 @@ define amdgpu_kernel void @lds_wg_rmw_add_acq_rel_single64(ptr addrspace(3) %p)
; GFX12-W64-NEXT: renamable $sgpr0 = S_MUL_I32 killed renamable $sgpr0, 3
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX12-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX12-W64-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
; GFX12-W64-NEXT: DS_ADD_U32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel (s32) on %ir.p.load, addrspace 3)
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_INV 8, implicit $exec
; GFX12-W64-NEXT: {{ $}}
; GFX12-W64-NEXT: bb.2 (%ir-block.16):
; GFX12-W64-NEXT: S_ENDPGM 0
@@ -2742,11 +2705,7 @@ define amdgpu_kernel void @lds_wg_cmpxchg_acq_rel_monotonic_single64(ptr addrspa
; GFX9-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
; GFX9-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: DS_CMPST_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $vgpr2, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.2, addrspace 3)
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: S_ENDPGM 0
;
; GFX942-LABEL: name: lds_wg_cmpxchg_acq_rel_monotonic_single64
@@ -2759,31 +2718,40 @@ define amdgpu_kernel void @lds_wg_cmpxchg_acq_rel_monotonic_single64(ptr addrspa
; GFX942-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX942-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
; GFX942-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: DS_CMPST_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $vgpr2, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.2, addrspace 3)
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: lds_wg_cmpxchg_acq_rel_monotonic_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX10-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX10-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
- ; GFX10-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
- ; GFX10-NEXT: DS_CMPST_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $vgpr2, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.2, addrspace 3)
- ; GFX10-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX10-NEXT: S_WAITCNT_lds_direct
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: lds_wg_cmpxchg_acq_rel_monotonic_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W32-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
+ ; GFX10-W32-NEXT: DS_CMPST_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $vgpr2, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.2, addrspace 3)
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX10-W64-LABEL: name: lds_wg_cmpxchg_acq_rel_monotonic_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: DS_CMPST_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $vgpr2, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.2, addrspace 3)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
;
; GFX12-W32-LABEL: name: lds_wg_cmpxchg_acq_rel_monotonic_single64
; GFX12-W32: bb.0 (%ir-block.0):
@@ -2814,14 +2782,7 @@ define amdgpu_kernel void @lds_wg_cmpxchg_acq_rel_monotonic_single64(ptr addrspa
; GFX12-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
; GFX12-W64-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
- ; GFX12-W64-NEXT: S_WAIT_BVHCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_SAMPLECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_STORECNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
; GFX12-W64-NEXT: DS_CMPSTORE_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $vgpr2, 0, 0, implicit $exec :: (load store syncscope("workgroup") acq_rel monotonic (s32) on %ir.2, addrspace 3)
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_INV 8, implicit $exec
; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: lds_wg_cmpxchg_acq_rel_monotonic_single64
@@ -2889,7 +2850,6 @@ define amdgpu_kernel void @lds_wg_cmpxchg_monotonic_acquire_single64(ptr addrspa
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
; GFX9-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
; GFX9-NEXT: DS_CMPST_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $vgpr2, 0, 0, implicit $exec :: (load store syncscope("workgroup") monotonic acquire (s32) on %ir.2, addrspace 3)
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
; GFX9-NEXT: S_ENDPGM 0
;
; GFX942-LABEL: name: lds_wg_cmpxchg_monotonic_acquire_single64
@@ -2903,23 +2863,35 @@ define amdgpu_kernel void @lds_wg_cmpxchg_monotonic_acquire_single64(ptr addrspa
; GFX942-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
; GFX942-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
; GFX942-NEXT: DS_CMPST_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $vgpr2, 0, 0, implicit $exec :: (load store syncscope("workgroup") monotonic acquire (s32) on %ir.2, addrspace 3)
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: lds_wg_cmpxchg_monotonic_acquire_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX10-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
- ; GFX10-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
- ; GFX10-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
- ; GFX10-NEXT: DS_CMPST_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $vgpr2, 0, 0, implicit $exec :: (load store syncscope("workgroup") monotonic acquire (s32) on %ir.2, addrspace 3)
- ; GFX10-NEXT: S_WAITCNT_soft .Lgkmcnt_0
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: lds_wg_cmpxchg_monotonic_acquire_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W32-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W32-NEXT: DS_CMPST_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $vgpr2, 0, 0, implicit $exec :: (load store syncscope("workgroup") monotonic acquire (s32) on %ir.2, addrspace 3)
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Lgkmcnt_0
+ ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX10-W64-LABEL: name: lds_wg_cmpxchg_monotonic_acquire_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W64-NEXT: $vgpr0 = V_MOV_B32_e32 killed $sgpr0, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
+ ; GFX10-W64-NEXT: DS_CMPST_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $vgpr2, 0, 0, implicit $exec :: (load store syncscope("workgroup") monotonic acquire (s32) on %ir.2, addrspace 3)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
;
; GFX12-W32-LABEL: name: lds_wg_cmpxchg_monotonic_acquire_single64
; GFX12-W32: bb.0 (%ir-block.0):
@@ -2946,8 +2918,6 @@ define amdgpu_kernel void @lds_wg_cmpxchg_monotonic_acquire_single64(ptr addrspa
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr2, implicit $exec, implicit $exec
; GFX12-W64-NEXT: $vgpr2 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $exec
; GFX12-W64-NEXT: DS_CMPSTORE_B32_gfx9 killed renamable $vgpr0, killed renamable $vgpr1, killed renamable $vgpr2, 0, 0, implicit $exec :: (load store syncscope("workgroup") monotonic acquire (s32) on %ir.2, addrspace 3)
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_INV 8, implicit $exec
; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: lds_wg_cmpxchg_monotonic_acquire_single64
@@ -2977,7 +2947,6 @@ define amdgpu_kernel void @flat_wg_ld_acquire_single64(ptr addrspace(0) %p) #1 {
; GFX9-NEXT: $vgpr0 = V_MOV_B32_e32 $sgpr0, implicit $exec, implicit $sgpr0_sgpr1
; GFX9-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $sgpr0_sgpr1, implicit $exec
; GFX9-NEXT: dead renamable $vgpr0 = FLAT_LOAD_DWORD killed renamable $vgpr0_vgpr1, 0, 0, implicit $exec, implicit $flat_scr :: (load syncscope("workgroup") acquire (s32) from %ir.p.load)
- ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
; GFX9-NEXT: S_ENDPGM 0
;
; GFX942-LABEL: name: flat_wg_ld_acquire_single64
@@ -2988,23 +2957,34 @@ define amdgpu_kernel void @flat_wg_ld_acquire_single64(ptr addrspace(0) %p) #1 {
; GFX942-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
; GFX942-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX942-NEXT: $vgpr0_vgpr1 = V_MOV_B64_e32 killed $sgpr0_sgpr1, implicit $exec, implicit $exec
- ; GFX942-NEXT: dead renamable $vgpr0 = FLAT_LOAD_DWORD killed renamable $vgpr0_vgpr1, 0, 1, implicit $exec, implicit $flat_scr :: (load syncscope("workgroup") acquire (s32) from %ir.p.load)
- ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
+ ; GFX942-NEXT: dead renamable $vgpr0 = FLAT_LOAD_DWORD killed renamable $vgpr0_vgpr1, 0, 0, implicit $exec, implicit $flat_scr :: (load syncscope("workgroup") acquire (s32) from %ir.p.load)
; GFX942-NEXT: S_ENDPGM 0
;
- ; GFX10-LABEL: name: flat_wg_ld_acquire_single64
- ; GFX10: bb.0 (%ir-block.0):
- ; GFX10-NEXT: liveins: $sgpr4_sgpr5
- ; GFX10-NEXT: {{ $}}
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
- ; GFX10-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
- ; GFX10-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
- ; GFX10-NEXT: $vgpr0 = V_MOV_B32_e32 $sgpr0, implicit $exec, implicit $sgpr0_sgpr1
- ; GFX10-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $sgpr0_sgpr1, implicit $exec
- ; GFX10-NEXT: dead renamable $vgpr0 = FLAT_LOAD_DWORD killed renamable $vgpr0_vgpr1, 0, 1, implicit $exec, implicit $flat_scr :: (load syncscope("workgroup") acquire (s32) from %ir.p.load)
- ; GFX10-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
- ; GFX10-NEXT: BUFFER_GL0_INV implicit $exec
- ; GFX10-NEXT: S_ENDPGM 0
+ ; GFX10-W32-LABEL: name: flat_wg_ld_acquire_single64
+ ; GFX10-W32: bb.0 (%ir-block.0):
+ ; GFX10-W32-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W32-NEXT: {{ $}}
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W32-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W32-NEXT: $vgpr0 = V_MOV_B32_e32 $sgpr0, implicit $exec, implicit $sgpr0_sgpr1
+ ; GFX10-W32-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $sgpr0_sgpr1, implicit $exec
+ ; GFX10-W32-NEXT: dead renamable $vgpr0 = FLAT_LOAD_DWORD killed renamable $vgpr0_vgpr1, 0, 1, implicit $exec, implicit $flat_scr :: (load syncscope("workgroup") acquire (s32) from %ir.p.load)
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: BUFFER_GL0_INV implicit $exec
+ ; GFX10-W32-NEXT: S_ENDPGM 0
+ ;
+ ; GFX10-W64-LABEL: name: flat_wg_ld_acquire_single64
+ ; GFX10-W64: bb.0 (%ir-block.0):
+ ; GFX10-W64-NEXT: liveins: $sgpr4_sgpr5
+ ; GFX10-W64-NEXT: {{ $}}
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION escape 0x0f, 0x04, 0x30, 0x36, 0xe9, 0x02
+ ; GFX10-W64-NEXT: frame-setup CFI_INSTRUCTION undefined $pc_reg
+ ; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
+ ; GFX10-W64-NEXT: $vgpr0 = V_MOV_B32_e32 $sgpr0, implicit $exec, implicit $sgpr0_sgpr1
+ ; GFX10-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $sgpr0_sgpr1, implicit $exec
+ ; GFX10-W64-NEXT: dead renamable $vgpr0 = FLAT_LOAD_DWORD killed renamable $vgpr0_vgpr1, 0, 0, implicit $exec, implicit $flat_scr :: (load syncscope("workgroup") acquire (s32) from %ir.p.load)
+ ; GFX10-W64-NEXT: S_ENDPGM 0
;
; GFX12-W32-LABEL: name: flat_wg_ld_acquire_single64
; GFX12-W32: bb.0 (%ir-block.0):
@@ -3029,10 +3009,7 @@ define amdgpu_kernel void @flat_wg_ld_acquire_single64(ptr addrspace(0) %p) #1 {
; GFX12-W64-NEXT: renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s64) from %ir.p.kernarg.offset, align 4, addrspace 4)
; GFX12-W64-NEXT: $vgpr0 = V_MOV_B32_e32 $sgpr0, implicit $exec, implicit $sgpr0_sgpr1
; GFX12-W64-NEXT: $vgpr1 = V_MOV_B32_e32 killed $sgpr1, implicit $exec, implicit $sgpr0_sgpr1, implicit $exec
- ; GFX12-W64-NEXT: dead renamable $vgpr0 = FLAT_LOAD_DWORD killed renamable $vgpr0_vgpr1, 0, 8, implicit $exec, implicit $flat_scr :: (load syncscope("workgroup") acquire (s32) from %ir.p.load)
- ; GFX12-W64-NEXT: S_WAIT_LOADCNT_soft 0
- ; GFX12-W64-NEXT: S_WAIT_DSCNT_soft 0
- ; GFX12-W64-NEXT: GLOBAL_INV 8, implicit $exec
+ ; GFX12-W64-NEXT: dead renamable $vgpr0 = FLAT_LOAD_DWORD killed renamable $vgpr0_vgpr1, 0, 0, implicit $exec, implicit $flat_scr :: (load syncscope("workgroup") acquire (s32) from %ir.p.load)
; GFX12-W64-NEXT: S_ENDPGM 0
;
; GFX1250-LABEL: name: flat_wg_ld_acquire_single64
>From 490c2fa85edbe9af70877b5e7323bedbafa4839e Mon Sep 17 00:00:00 2001
From: Zach Goldthorpe <Zach.Goldthorpe at amd.com>
Date: Wed, 16 Sep 2026 09:58:15 -0500
Subject: [PATCH 2/4] Prohibit demotion if function contains LDSDMA
---
llvm/docs/AMDGPUUsage.rst | 23 +++++++++++--------
llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp | 20 ++++++++++++----
llvm/test/CodeGen/AMDGPU/lds-dma-war-async.ll | 6 +++--
llvm/test/CodeGen/AMDGPU/lds-dma-war.ll | 7 ++++--
...zer-single-wave-workgroup-lds-dma-async.ll | 3 +++
...legalizer-single-wave-workgroup-lds-dma.ll | 10 ++++++++
6 files changed, 50 insertions(+), 19 deletions(-)
diff --git a/llvm/docs/AMDGPUUsage.rst b/llvm/docs/AMDGPUUsage.rst
index dedbec9c6a415c..122a064959acbf 100644
--- a/llvm/docs/AMDGPUUsage.rst
+++ b/llvm/docs/AMDGPUUsage.rst
@@ -7546,16 +7546,19 @@ A memory synchronization scope wider than work-group is not meaningful for the
group (LDS) address space and is treated as work-group.
When a work-group's maximum flat work-group size does not exceed the wavefront
-size, the work-group fits within a single wavefront. In this case, LLVM
-``workgroup`` synchronization scope is equivalent to ``wavefront`` scope.
-
-If the compiler can determine this bound (e.g., via ``amdgpu-flat-work-group-size``),
-the AMDGPU backend optimizes ``workgroup`` scope operations by lowering them to
-``wavefront``-scoped machine instructions.
-
-It applies to atomic ``load``, ``store``, ``atomicrmw``, and ``cmpxchg``
-instructions, and to ``fence`` instructions, when they use synchronizing memory
-orderings (``acquire``, ``release``, ``acq_rel``, or ``seq_cst``).
+size, the work-group fits within a single wavefront. So long as no LDS DMA
+operations occur, the LLVM ``workgroup`` synchronization scope is equivalent to
+its ``wavefront`` scope.
+
+If the compiler can determine these conditions (e.g., determining the
+work-group size via ``amdgpu-flat-work-group-size`` and detecting no LDS DMA
+operations), the AMDGPU backend optimizes ``workgroup`` scope operations by
+lowering them to ``wavefront``-scoped machine instructions.
+
+This optimization applies to atomic ``load``, ``store``, ``atomicrmw``, and
+``cmpxchg`` instructions, and to ``fence`` instructions, when they use
+synchronizing memory orderings (``acquire``, ``release``, ``acq_rel``, or
+``seq_cst``).
The memory model does not support the region address space which is treated as
non-atomic.
diff --git a/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp b/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
index b6884f2a51268b..f38d7f60b6fbc7 100644
--- a/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
+++ b/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
@@ -317,7 +317,7 @@ class SIMemOpAccess final {
/// Construct class to support accessing the machine memory operands
/// of instructions.
SIMemOpAccess(const AMDGPUMachineModuleInfo &MMI, const GCNSubtarget &ST,
- const Function &F);
+ const MachineFunction &MF);
/// \returns Load info if \p MI is a load operation, "std::nullopt" otherwise.
std::optional<SIMemOpInfo>
@@ -838,13 +838,23 @@ SIAtomicAddrSpace SIMemOpAccess::toSIAtomicAddrSpace(unsigned AS) const {
return SIAtomicAddrSpace::OTHER;
}
+/// returns true if any instruction in \p MF accesses LDS through DMA.
+static bool containsLDSDMA(const MachineFunction &MF) {
+ return any_of(MF, [](const MachineBasicBlock &MBB) {
+ return any_of(MBB.instrs(), [](const MachineInstr &MI) {
+ return SIInstrInfo::isLDSDMA(MI);
+ });
+ });
+}
+
// TODO: Consider moving single-wave workgroup->wavefront scope relaxation to an
// IR pass (and extending it to other scoped operations), so middle-end
// optimizations see wavefront scope earlier.
SIMemOpAccess::SIMemOpAccess(const AMDGPUMachineModuleInfo &MMI_,
- const GCNSubtarget &ST, const Function &F)
- : MMI(&MMI_), ST(ST),
- CanDemoteWorkgroupToWavefront(ST.isSingleWavefrontWorkgroup(F)) {}
+ const GCNSubtarget &ST, const MachineFunction &MF)
+ : MMI(&MMI_), ST(ST), CanDemoteWorkgroupToWavefront(
+ ST.isSingleWavefrontWorkgroup(MF.getFunction()) &&
+ !containsLDSDMA(MF)) {}
std::optional<SIMemOpInfo> SIMemOpAccess::constructFromMIWithMMO(
const MachineBasicBlock::iterator &MI) const {
@@ -2604,7 +2614,7 @@ bool SIMemoryLegalizer::run(MachineFunction &MF) {
const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
const Function &F = MF.getFunction();
- SIMemOpAccess MOA(MMI.getObjFileInfo<AMDGPUMachineModuleInfo>(), ST, F);
+ SIMemOpAccess MOA(MMI.getObjFileInfo<AMDGPUMachineModuleInfo>(), ST, MF);
bool TgSplit = ST.hasTgSplitSupport() && AMDGPU::isTgSplitEnabled(F);
CC = SICacheControl::create(ST, TgSplit);
diff --git a/llvm/test/CodeGen/AMDGPU/lds-dma-war-async.ll b/llvm/test/CodeGen/AMDGPU/lds-dma-war-async.ll
index 8be725c7bef2cc..c89f41f8100608 100644
--- a/llvm/test/CodeGen/AMDGPU/lds-dma-war-async.ll
+++ b/llvm/test/CodeGen/AMDGPU/lds-dma-war-async.ll
@@ -45,8 +45,8 @@ define i32 @lds_then_dma.global.wg(ptr addrspace(1) %g, ptr addrspace(3) inreg %
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: v_mov_b32_e32 v3, s0
; CHECK-NEXT: ds_load_b32 v2, v3
-; CHECK-NEXT: global_load_async_to_lds_b32 v3, v[0:1], off
; CHECK-NEXT: s_wait_dscnt 0x0
+; CHECK-NEXT: global_load_async_to_lds_b32 v3, v[0:1], off
; CHECK-NEXT: v_mov_b32_e32 v0, v2
; CHECK-NEXT: s_set_pc_i64 s[30:31]
%v = load i32, ptr addrspace(3) %lds, align 4
@@ -136,6 +136,7 @@ define i32 @lds_then_dma.global.wg.partial(ptr addrspace(1) %g, ptr addrspace(3)
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v4, s1
; CHECK-NEXT: ds_load_b32 v3, v2
+; CHECK-NEXT: s_wait_dscnt 0x0
; CHECK-NEXT: ds_load_b32 v4, v4
; CHECK-NEXT: global_load_async_to_lds_b32 v2, v[0:1], off
; CHECK-NEXT: s_wait_dscnt 0x0
@@ -230,8 +231,8 @@ define i32 @lds_then_dma.tensor.wg(<4 x i32> inreg %D0, <8 x i32> inreg %D1, ptr
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: v_mov_b32_e32 v0, s24
; CHECK-NEXT: ds_load_b32 v0, v0
-; CHECK-NEXT: tensor_load_to_lds s[0:3], s[16:23]
; CHECK-NEXT: s_wait_dscnt 0x0
+; CHECK-NEXT: tensor_load_to_lds s[0:3], s[16:23]
; CHECK-NEXT: s_set_pc_i64 s[30:31]
%v = load i32, ptr addrspace(3) %lds, align 4
fence syncscope("workgroup") release, !mmra !0
@@ -318,6 +319,7 @@ define i32 @lds_then_dma.tensor.wg.partial(<4 x i32> inreg %D0, <8 x i32> inreg
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: v_dual_mov_b32 v0, s24 :: v_dual_mov_b32 v1, s25
; CHECK-NEXT: ds_load_b32 v0, v0
+; CHECK-NEXT: s_wait_dscnt 0x0
; CHECK-NEXT: ds_load_b32 v1, v1
; CHECK-NEXT: tensor_load_to_lds s[0:3], s[16:23]
; CHECK-NEXT: s_wait_dscnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/lds-dma-war.ll b/llvm/test/CodeGen/AMDGPU/lds-dma-war.ll
index 6ee1b2ed83b808..77b44f304abfed 100644
--- a/llvm/test/CodeGen/AMDGPU/lds-dma-war.ll
+++ b/llvm/test/CodeGen/AMDGPU/lds-dma-war.ll
@@ -47,8 +47,8 @@ define i32 @lds_then_dma.global.wg(ptr addrspace(1) %g, ptr addrspace(3) inreg %
; CHECK-NEXT: v_mov_b32_e32 v2, s0
; CHECK-NEXT: s_mov_b32 m0, s0
; CHECK-NEXT: ds_read_b32 v2, v2
-; CHECK-NEXT: global_load_lds_dword v[0:1], off
; CHECK-NEXT: s_waitcnt lgkmcnt(0)
+; CHECK-NEXT: global_load_lds_dword v[0:1], off
; CHECK-NEXT: v_mov_b32_e32 v0, v2
; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: s_setpc_b64 s[30:31]
@@ -146,6 +146,7 @@ define i32 @lds_then_dma.global.wg.partial(ptr addrspace(1) %g, ptr addrspace(3)
; CHECK-NEXT: v_mov_b32_e32 v2, s0
; CHECK-NEXT: v_mov_b32_e32 v3, s1
; CHECK-NEXT: ds_read_b32 v2, v2
+; CHECK-NEXT: s_waitcnt lgkmcnt(0)
; CHECK-NEXT: ds_read_b32 v3, v3
; CHECK-NEXT: global_load_lds_dword v[0:1], off
; CHECK-NEXT: s_waitcnt lgkmcnt(0)
@@ -245,8 +246,9 @@ define i32 @lds_then_dma.buffer.wg(<4 x i32> inreg %rsrc, ptr addrspace(3) inreg
; CHECK-NEXT: v_mov_b32_e32 v0, s16
; CHECK-NEXT: s_mov_b32 m0, s16
; CHECK-NEXT: ds_read_b32 v0, v0
+; CHECK-NEXT: s_waitcnt lgkmcnt(0)
; CHECK-NEXT: buffer_load_dword off, s[0:3], 0 lds
-; CHECK-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: s_setpc_b64 s[30:31]
%v = load i32, ptr addrspace(3) %lds, align 4
fence syncscope("workgroup") release, !mmra !0
@@ -340,6 +342,7 @@ define i32 @lds_then_dma.buffer.wg.partial(<4 x i32> inreg %rsrc, ptr addrspace(
; CHECK-NEXT: v_mov_b32_e32 v0, s16
; CHECK-NEXT: v_mov_b32_e32 v1, s17
; CHECK-NEXT: ds_read_b32 v0, v0
+; CHECK-NEXT: s_waitcnt lgkmcnt(0)
; CHECK-NEXT: ds_read_b32 v1, v1
; CHECK-NEXT: buffer_load_dword off, s[0:3], 0 lds
; CHECK-NEXT: s_waitcnt lgkmcnt(0)
diff --git a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma-async.ll b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma-async.ll
index 826436c2ca5bc8..f4c397c39dd4ff 100644
--- a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma-async.ll
+++ b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma-async.ll
@@ -32,6 +32,9 @@ define amdgpu_kernel void @lds_async_dma_wg_fence_release_single32(ptr addrspace
; GFX1250-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2 = S_LOAD_DWORDX3_IMM_ec killed renamable $sgpr4_sgpr5, 36, 32 :: (dereferenceable invariant load (s96) from %ir.g.kernarg.offset, align 4, addrspace 4)
; GFX1250-NEXT: renamable $vgpr0, $vgpr1 = V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1250 0, killed $sgpr2, implicit $exec, implicit $exec, implicit $exec, implicit $exec
; GFX1250-NEXT: GLOBAL_LOAD_ASYNC_TO_LDS_B32_SADDR killed $vgpr1, killed $sgpr0_sgpr1, killed $vgpr0, 0, 0, implicit-def dead $asynccnt, implicit $exec, implicit $asynccnt :: (load (s32) from %ir.g.load, align 1, addrspace 1), (store (s32) into %ir.lds.load, align 1, addrspace 3)
+ ; GFX1250-NEXT: S_WAIT_LOADCNT_soft 0
+ ; GFX1250-NEXT: S_WAIT_STORECNT_soft 0
+ ; GFX1250-NEXT: S_WAIT_DSCNT_soft 0
; GFX1250-NEXT: S_ENDPGM 0
call void @llvm.amdgcn.global.load.async.to.lds.b32(ptr addrspace(1) %g, ptr addrspace(3) %lds, i32 0, i32 0)
fence syncscope("workgroup") release
diff --git a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma.ll b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma.ll
index 5944db56e6a3b3..707910ce005dd8 100644
--- a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma.ll
+++ b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma.ll
@@ -125,6 +125,8 @@ define amdgpu_kernel void @lds_dma_wg_fence_release_single32(ptr addrspace(8) %r
; GFX9-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.rsrc.kernarg.offset, align 4, addrspace 4)
; GFX9-NEXT: $m0 = S_MOV_B32 killed $sgpr6
; GFX9-NEXT: BUFFER_LOAD_DWORD_LDS_OFFSET killed $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, 0, implicit $exec, implicit $m0 :: (dereferenceable load (s32) from %ir.rsrc.load, align 1, addrspace 8), (dereferenceable store (s2048) into %ir.lds.load, align 1, addrspace 3)
+ ; GFX9-NEXT: S_WAITCNT_soft .Lgkmcnt_0
+ ; GFX9-NEXT: S_WAITCNT_lds_direct
; GFX9-NEXT: S_ENDPGM 0
;
; GFX942-LABEL: name: lds_dma_wg_fence_release_single32
@@ -137,6 +139,8 @@ define amdgpu_kernel void @lds_dma_wg_fence_release_single32(ptr addrspace(8) %r
; GFX942-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.rsrc.kernarg.offset, align 4, addrspace 4)
; GFX942-NEXT: $m0 = S_MOV_B32 killed $sgpr6
; GFX942-NEXT: BUFFER_LOAD_DWORD_LDS_OFFSET killed $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, 0, implicit $exec, implicit $m0 :: (dereferenceable load (s32) from %ir.rsrc.load, align 1, addrspace 8), (dereferenceable store (s2048) into %ir.lds.load, align 1, addrspace 3)
+ ; GFX942-NEXT: S_WAITCNT_soft .Lgkmcnt_0
+ ; GFX942-NEXT: S_WAITCNT_lds_direct
; GFX942-NEXT: S_ENDPGM 0
;
; GFX10-W32-LABEL: name: lds_dma_wg_fence_release_single32
@@ -149,6 +153,9 @@ define amdgpu_kernel void @lds_dma_wg_fence_release_single32(ptr addrspace(8) %r
; GFX10-W32-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.rsrc.kernarg.offset, align 4, addrspace 4)
; GFX10-W32-NEXT: $m0 = S_MOV_B32 killed $sgpr6
; GFX10-W32-NEXT: BUFFER_LOAD_DWORD_LDS_OFFSET killed $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, 0, implicit $exec, implicit $m0 :: (dereferenceable load (s32) from %ir.rsrc.load, align 1, addrspace 8), (dereferenceable store (s1024) into %ir.lds.load, align 1, addrspace 3)
+ ; GFX10-W32-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W32-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W32-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
; GFX10-W32-NEXT: S_ENDPGM 0
;
; GFX10-W64-LABEL: name: lds_dma_wg_fence_release_single32
@@ -161,6 +168,9 @@ define amdgpu_kernel void @lds_dma_wg_fence_release_single32(ptr addrspace(8) %r
; GFX10-W64-NEXT: early-clobber renamable $sgpr0_sgpr1_sgpr2_sgpr3 = S_LOAD_DWORDX4_IMM_ec killed renamable $sgpr4_sgpr5, 36, 0 :: (dereferenceable invariant load (s128) from %ir.rsrc.kernarg.offset, align 4, addrspace 4)
; GFX10-W64-NEXT: $m0 = S_MOV_B32 killed $sgpr6
; GFX10-W64-NEXT: BUFFER_LOAD_DWORD_LDS_OFFSET killed $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, 0, implicit $exec, implicit $m0 :: (dereferenceable load (s32) from %ir.rsrc.load, align 1, addrspace 8), (dereferenceable store (s2048) into %ir.lds.load, align 1, addrspace 3)
+ ; GFX10-W64-NEXT: S_WAITCNT_soft .Vmcnt_0_Lgkmcnt_0
+ ; GFX10-W64-NEXT: S_WAITCNT_lds_direct
+ ; GFX10-W64-NEXT: S_WAITCNT_VSCNT_soft undef $sgpr_null, 0
; GFX10-W64-NEXT: S_ENDPGM 0
call void @llvm.amdgcn.raw.ptr.buffer.load.lds(ptr addrspace(8) %rsrc, ptr addrspace(3) %lds, i32 4, i32 0, i32 0, i32 0, i32 0)
fence syncscope("workgroup") release
>From 2eb6ab8938a87c51814004df3465b18021da78ba Mon Sep 17 00:00:00 2001
From: Zach Goldthorpe <Zach.Goldthorpe at amd.com>
Date: Thu, 17 Sep 2026 14:47:36 -0500
Subject: [PATCH 3/4] Remove unnecessary condition on ordering strength
---
llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp | 9 +--
...-legalizer-single-wave-workgroup-memops.ll | 72 ++++++++++++-------
2 files changed, 47 insertions(+), 34 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp b/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
index f38d7f60b6fbc7..6817781871edc3 100644
--- a/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
+++ b/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
@@ -209,15 +209,8 @@ class SIMemOpInfo final {
if (this->Scope == SIAtomicScope::CLUSTER && !ST.hasClusters())
this->Scope = SIAtomicScope::AGENT;
- // When max flat work-group size is at most the wavefront size, the
- // work-group fits in a single wave, so LLVM workgroup scope matches
- // wavefront scope. Demote workgroup → wavefront here for fences and for
- // atomics with ordering stronger than monotonic.
if (CanDemoteWorkgroupToWavefront &&
- this->Scope == SIAtomicScope::WORKGROUP &&
- (llvm::isStrongerThan(this->Ordering, AtomicOrdering::Monotonic) ||
- llvm::isStrongerThan(this->FailureOrdering,
- AtomicOrdering::Monotonic)))
+ this->Scope == SIAtomicScope::WORKGROUP)
this->Scope = SIAtomicScope::WAVEFRONT;
}
diff --git a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-memops.ll b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-memops.ll
index 8341365d40bb0b..5c91713507e205 100644
--- a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-memops.ll
+++ b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-memops.ll
@@ -575,7 +575,7 @@ define amdgpu_kernel void @wg_ld_monotonic_single32(ptr addrspace(1) %p) #0 {
; GFX942-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX942-NEXT: v_mov_b32_e32 v0, 0
; GFX942-NEXT: s_waitcnt lgkmcnt(0)
-; GFX942-NEXT: global_load_dword v0, v0, s[0:1] sc0
+; GFX942-NEXT: global_load_dword v0, v0, s[0:1]
; GFX942-NEXT: s_endpgm
;
; GFX10-LABEL: wg_ld_monotonic_single32:
@@ -585,7 +585,7 @@ define amdgpu_kernel void @wg_ld_monotonic_single32(ptr addrspace(1) %p) #0 {
; GFX10-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX10-NEXT: v_mov_b32_e32 v0, 0
; GFX10-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10-NEXT: global_load_dword v0, v0, s[0:1] glc
+; GFX10-NEXT: global_load_dword v0, v0, s[0:1]
; GFX10-NEXT: s_endpgm
;
; GFX12-LABEL: wg_ld_monotonic_single32:
@@ -595,7 +595,7 @@ define amdgpu_kernel void @wg_ld_monotonic_single32(ptr addrspace(1) %p) #0 {
; GFX12-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: global_load_b32 v0, v0, s[0:1] scope:SCOPE_SE
+; GFX12-NEXT: global_load_b32 v0, v0, s[0:1]
; GFX12-NEXT: s_endpgm
;
; GFX1250-LABEL: wg_ld_monotonic_single32:
@@ -633,28 +633,48 @@ define amdgpu_kernel void @wg_ld_monotonic_single64(ptr addrspace(1) %p) #1 {
; GFX942-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX942-NEXT: v_mov_b32_e32 v0, 0
; GFX942-NEXT: s_waitcnt lgkmcnt(0)
-; GFX942-NEXT: global_load_dword v0, v0, s[0:1] sc0
+; GFX942-NEXT: global_load_dword v0, v0, s[0:1]
; GFX942-NEXT: s_endpgm
;
-; GFX10-LABEL: wg_ld_monotonic_single64:
-; GFX10: ; %bb.0:
-; GFX10-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
-; GFX10-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
-; GFX10-NEXT: v_mov_b32_e32 v0, 0
-; GFX10-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10-NEXT: global_load_dword v0, v0, s[0:1] glc
-; GFX10-NEXT: s_endpgm
+; GFX10-W32-LABEL: wg_ld_monotonic_single64:
+; GFX10-W32: ; %bb.0:
+; GFX10-W32-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
+; GFX10-W32-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-W32-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
+; GFX10-W32-NEXT: v_mov_b32_e32 v0, 0
+; GFX10-W32-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-W32-NEXT: global_load_dword v0, v0, s[0:1] glc
+; GFX10-W32-NEXT: s_endpgm
;
-; GFX12-LABEL: wg_ld_monotonic_single64:
-; GFX12: ; %bb.0:
-; GFX12-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX12-NEXT: v_mov_b32_e32 v0, 0
-; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: global_load_b32 v0, v0, s[0:1] scope:SCOPE_SE
-; GFX12-NEXT: s_endpgm
+; GFX10-W64-LABEL: wg_ld_monotonic_single64:
+; GFX10-W64: ; %bb.0:
+; GFX10-W64-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
+; GFX10-W64-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-W64-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
+; GFX10-W64-NEXT: v_mov_b32_e32 v0, 0
+; GFX10-W64-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-W64-NEXT: global_load_dword v0, v0, s[0:1]
+; GFX10-W64-NEXT: s_endpgm
+;
+; GFX12-W32-LABEL: wg_ld_monotonic_single64:
+; GFX12-W32: ; %bb.0:
+; GFX12-W32-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX12-W32-NEXT: s_wait_kmcnt 0x0
+; GFX12-W32-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX12-W32-NEXT: v_mov_b32_e32 v0, 0
+; GFX12-W32-NEXT: s_wait_kmcnt 0x0
+; GFX12-W32-NEXT: global_load_b32 v0, v0, s[0:1] scope:SCOPE_SE
+; GFX12-W32-NEXT: s_endpgm
+;
+; GFX12-W64-LABEL: wg_ld_monotonic_single64:
+; GFX12-W64: ; %bb.0:
+; GFX12-W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX12-W64-NEXT: s_wait_kmcnt 0x0
+; GFX12-W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX12-W64-NEXT: v_mov_b32_e32 v0, 0
+; GFX12-W64-NEXT: s_wait_kmcnt 0x0
+; GFX12-W64-NEXT: global_load_b32 v0, v0, s[0:1]
+; GFX12-W64-NEXT: s_endpgm
;
; GFX1250-LABEL: wg_ld_monotonic_single64:
; GFX1250: ; %bb.0:
@@ -1155,7 +1175,7 @@ define amdgpu_kernel void @wg_st_monotonic_single32(ptr addrspace(1) %p, i32 %x)
; GFX942-NEXT: s_load_dword s2, s[4:5], 0x2c
; GFX942-NEXT: s_waitcnt lgkmcnt(0)
; GFX942-NEXT: v_mov_b32_e32 v1, s2
-; GFX942-NEXT: global_store_dword v0, v1, s[0:1] sc0
+; GFX942-NEXT: global_store_dword v0, v1, s[0:1]
; GFX942-NEXT: s_endpgm
;
; GFX10-LABEL: wg_st_monotonic_single32:
@@ -1183,7 +1203,7 @@ define amdgpu_kernel void @wg_st_monotonic_single32(ptr addrspace(1) %p, i32 %x)
; GFX12-NEXT: s_load_b32 s2, s[4:5], 0x2c
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: v_mov_b32_e32 v1, s2
-; GFX12-NEXT: global_store_b32 v0, v1, s[0:1] scope:SCOPE_SE
+; GFX12-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX12-NEXT: s_endpgm
;
; GFX1250-LABEL: wg_st_monotonic_single32:
@@ -1687,7 +1707,7 @@ define amdgpu_kernel void @wg_rmw_add_monotonic_single32(ptr addrspace(1) %p) #0
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: v_mov_b32_e32 v1, 7
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: global_atomic_add_u32 v0, v1, s[0:1] scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_add_u32 v0, v1, s[0:1]
; GFX12-NEXT: s_endpgm
;
; GFX1250-LABEL: wg_rmw_add_monotonic_single32:
@@ -2291,7 +2311,7 @@ define amdgpu_kernel void @wg_cmpxchg_monotonic_monotonic_single32(ptr addrspace
; GFX12-NEXT: v_mov_b32_e32 v3, s2
; GFX12-NEXT: ; kill: def $vgpr1 killed $vgpr1 def $vgpr1_vgpr2 killed $exec
; GFX12-NEXT: v_mov_b32_e32 v2, v3
-; GFX12-NEXT: global_atomic_cmpswap_b32 v0, v[1:2], s[0:1] scope:SCOPE_SE
+; GFX12-NEXT: global_atomic_cmpswap_b32 v0, v[1:2], s[0:1]
; GFX12-NEXT: s_endpgm
;
; GFX1250-LABEL: wg_cmpxchg_monotonic_monotonic_single32:
>From fcb28f9a24b9f1b70873c0d3dbd1cf340f9257f5 Mon Sep 17 00:00:00 2001
From: Zach Goldthorpe <Zach.Goldthorpe at amd.com>
Date: Thu, 17 Sep 2026 16:25:15 -0500
Subject: [PATCH 4/4] Check for async operations through attributor
---
llvm/docs/AMDGPUUsage.rst | 13 +-
llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp | 98 ++-
llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp | 16 +-
.../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 39 +
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 3 +
.../AMDGPU/addrspacecast-constantexpr.ll | 4 +-
...mdgpu-attributor-flat-scratch-init-asan.ll | 2 +-
.../amdgpu-attributor-min-agpr-alloc.ll | 178 ++---
.../AMDGPU/amdgpu-attributor-no-async.ll | 286 ++++++++
...amdgpu-attributor-nocallback-intrinsics.ll | 27 +-
.../AMDGPU/amdgpu-attributor-trap-leaf.ll | 11 +-
...-kernel-features-hsa-call-addrspacecast.ll | 8 +-
.../annotate-kernel-features-hsa-call.ll | 61 +-
.../AMDGPU/annotate-kernel-features-hsa.ll | 26 +-
.../AMDGPU/annotate-kernel-features.ll | 18 +-
...utor-flatscratchinit-undefined-behavior.ll | 4 +-
.../AMDGPU/attributor-flatscratchinit.ll | 18 +-
llvm/test/CodeGen/AMDGPU/attributor-wwm.ll | 4 +-
.../CodeGen/AMDGPU/direct-indirect-call.ll | 2 +-
.../AMDGPU/duplicate-attribute-indirect.ll | 2 +-
.../test/CodeGen/AMDGPU/flat-saddr-atomics.ll | 192 ++---
.../CodeGen/AMDGPU/global-saddr-atomics.ll | 672 +++++++++++++-----
.../AMDGPU/implicitarg-offset-attributes.ll | 26 +-
.../indirect-call-set-from-other-function.ll | 2 +-
...120256-annotate-constexpr-addrspacecast.ll | 4 +-
...zer-single-wave-workgroup-lds-dma-async.ll | 5 +-
...legalizer-single-wave-workgroup-lds-dma.ll | 7 +-
...-legalizer-single-wave-workgroup-memops.ll | 6 +-
.../AMDGPU/propagate-flat-work-group-size.ll | 18 +-
.../CodeGen/AMDGPU/propagate-waves-per-eu.ll | 42 +-
.../AMDGPU/recursive_global_initializer.ll | 2 +-
.../AMDGPU/remove-no-kernel-id-attribute.ll | 10 +-
.../CodeGen/AMDGPU/simple-indirect-call-2.ll | 6 +-
.../CodeGen/AMDGPU/simple-indirect-call.ll | 2 +-
.../uniform-work-group-attribute-missing.ll | 4 +-
.../AMDGPU/uniform-work-group-multistep.ll | 4 +-
...niform-work-group-nested-function-calls.ll | 4 +-
...ork-group-prevent-attribute-propagation.ll | 4 +-
.../uniform-work-group-propagate-attribute.ll | 4 +-
.../uniform-work-group-recursion-test.ll | 6 +-
.../CodeGen/AMDGPU/uniform-work-group-test.ll | 2 +-
41 files changed, 1311 insertions(+), 531 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/amdgpu-attributor-no-async.ll
diff --git a/llvm/docs/AMDGPUUsage.rst b/llvm/docs/AMDGPUUsage.rst
index 122a064959acbf..cb8f139aa82a99 100644
--- a/llvm/docs/AMDGPUUsage.rst
+++ b/llvm/docs/AMDGPUUsage.rst
@@ -2711,6 +2711,9 @@ The AMDGPU backend supports the following LLVM IR attributes.
kernel argument that holds the completion action pointer. If this
attribute is absent, then the amdgpu-no-implicitarg-ptr is also removed.
+ "amdgpu-no-async" Indicates the function does not execute any asynchronous operations
+ (LDS DMA, ASYNC, TENSOR).
+
"amdgpu-tg-split" Enable threadgroup split execution mode for the function. This must be
consistently set (or unset) for all reachable functions. This is only
relevant on targets with the `tgsplit-support` feature.
@@ -7546,14 +7549,14 @@ A memory synchronization scope wider than work-group is not meaningful for the
group (LDS) address space and is treated as work-group.
When a work-group's maximum flat work-group size does not exceed the wavefront
-size, the work-group fits within a single wavefront. So long as no LDS DMA
+size, the work-group fits within a single wavefront. So long as no asynchronous
operations occur, the LLVM ``workgroup`` synchronization scope is equivalent to
its ``wavefront`` scope.
-If the compiler can determine these conditions (e.g., determining the
-work-group size via ``amdgpu-flat-work-group-size`` and detecting no LDS DMA
-operations), the AMDGPU backend optimizes ``workgroup`` scope operations by
-lowering them to ``wavefront``-scoped machine instructions.
+If the compiler can determine these conditions (e.g., through the function attributes
+``amdgpu-flat-work-group-size`` and ``amdgpu-no-async``), the AMDGPU backend
+optimizes ``workgroup`` scope operations by lowering them to
+``wavefront``-scoped machine instructions.
This optimization applies to atomic ``load``, ``store``, ``atomicrmw``, and
``cmpxchg`` instructions, and to ``fence`` instructions, when they use
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
index 2ed1be7c2ade05..0e38a38e7d1c7c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
@@ -1401,6 +1401,101 @@ struct AAAMDGPUMinAGPRAlloc
const char AAAMDGPUMinAGPRAlloc::ID = 0;
+/// Deduce the function attribute "amdgpu-no-async"
+struct AAAMDGPUNoAsync : public StateWrapper<BooleanState, AbstractAttribute> {
+ using Base = StateWrapper<BooleanState, AbstractAttribute>;
+ AAAMDGPUNoAsync(const IRPosition &IRP, Attributor &A) : Base(IRP) {}
+
+ /// Create an abstract attribute view for the position \p IRP.
+ static AAAMDGPUNoAsync &createForPosition(const IRPosition &IRP,
+ Attributor &A) {
+ if (IRP.getPositionKind() == IRPosition::IRP_FUNCTION)
+ return *new (A.Allocator) AAAMDGPUNoAsync(IRP, A);
+ llvm_unreachable("AAAMDGPUNoAsync is only valid for function position");
+ }
+
+ void initialize(Attributor &A) override {
+ const Function *F = getAssociatedFunction();
+ if (F->hasFnAttribute("amdgpu-no-async")) {
+ indicateOptimisticFixpoint();
+ return;
+ }
+
+ if (F->isDeclaration())
+ indicatePessimisticFixpoint();
+ }
+
+ ChangeStatus updateImpl(Attributor &A) override {
+ auto CheckNoAsync = [&](Instruction &I) {
+ const auto &CB = cast<CallBase>(I);
+
+ // Inline assembly may hold any instruction, including an asynchronous
+ // operation.
+ if (isa<InlineAsm>(CB.getCalledOperand()))
+ return false;
+
+ Intrinsic::ID IID = CB.getIntrinsicID();
+ if (IID != Intrinsic::not_intrinsic) {
+ // Assume !nocallback intrinsics may call a function which issues an
+ // asynchronous operation.
+ return !AMDGPU::isAsyncIntrinsic(IID) &&
+ CB.hasFnAttr(Attribute::NoCallback);
+ }
+
+ const auto *CBEdges = A.getAAFor<AACallEdges>(
+ *this, IRPosition::callsite_function(CB), DepClassTy::REQUIRED);
+ if (!CBEdges || CBEdges->hasUnknownCallee())
+ return false;
+
+ for (const Function *PossibleCallee : CBEdges->getOptimisticEdges()) {
+ const auto *CalleeInfo = A.getAAFor<AAAMDGPUNoAsync>(
+ *this, IRPosition::function(*PossibleCallee), DepClassTy::REQUIRED);
+ if (!CalleeInfo || !CalleeInfo->getAssumed())
+ return false;
+ }
+
+ return true;
+ };
+
+ bool UsedAssumedInformation = false;
+ if (!A.checkForAllCallLikeInstructions(CheckNoAsync, *this,
+ UsedAssumedInformation))
+ return indicatePessimisticFixpoint();
+
+ return ChangeStatus::UNCHANGED;
+ }
+
+ ChangeStatus manifest(Attributor &A) override {
+ if (!getAssumed())
+ return ChangeStatus::UNCHANGED;
+
+ LLVMContext &Ctx = getAssociatedFunction()->getContext();
+ return A.manifestAttrs(getIRPosition(),
+ {Attribute::get(Ctx, "amdgpu-no-async")});
+ }
+
+ bool isValidState() const override { return true; }
+
+ const std::string getAsStr(Attributor *A) const override {
+ return "AMDGPUNoAsync[" + std::to_string(getAssumed()) + "]";
+ }
+
+ void trackStatistics() const override {}
+
+ StringRef getName() const override { return "AAAMDGPUNoAsync"; }
+ const char *getIdAddr() const override { return &ID; }
+
+ /// This function should return true if the type of the \p AA is
+ /// AAAMDGPUNoAsync
+ static bool classof(const AbstractAttribute *AA) {
+ return (AA->getIdAddr() == &ID);
+ }
+
+ static const char ID;
+};
+
+const char AAAMDGPUNoAsync::ID = 0;
+
/// An abstract attribute to propagate the function attribute
/// "amdgpu-cluster-dims" from kernel entry functions to device functions.
struct AAAMDGPUClusterDims
@@ -1567,7 +1662,7 @@ static bool runImpl(SetVector<Function *> &Functions, bool IsModulePass,
&AAAMDGPUMinAGPRAlloc::ID, &AACallEdges::ID, &AAPointerInfo::ID,
&AAPotentialConstantValues::ID, &AAUnderlyingObjects::ID,
&AANoAliasAddrSpace::ID, &AAAddressSpace::ID, &AAIndirectCallInfo::ID,
- &AAAMDGPUClusterDims::ID, &AAAlign::ID});
+ &AAAMDGPUClusterDims::ID, &AAAMDGPUNoAsync::ID, &AAAlign::ID});
AttributorConfig AC(CGUpdater);
AC.IsClosedWorldModule = Options.IsClosedWorld;
@@ -1599,6 +1694,7 @@ static bool runImpl(SetVector<Function *> &Functions, bool IsModulePass,
A.getOrCreateAAFor<AAAMDAttributes>(IRPosition::function(*F));
A.getOrCreateAAFor<AAUniformWorkGroupSize>(IRPosition::function(*F));
A.getOrCreateAAFor<AAAMDMaxNumWorkgroups>(IRPosition::function(*F));
+ A.getOrCreateAAFor<AAAMDGPUNoAsync>(IRPosition::function(*F));
CallingConv::ID CC = F->getCallingConv();
if (!AMDGPU::isEntryFunctionCC(CC)) {
A.getOrCreateAAFor<AAAMDFlatWorkGroupSize>(IRPosition::function(*F));
diff --git a/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp b/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
index 6817781871edc3..7f182520c08e79 100644
--- a/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
+++ b/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
@@ -831,23 +831,15 @@ SIAtomicAddrSpace SIMemOpAccess::toSIAtomicAddrSpace(unsigned AS) const {
return SIAtomicAddrSpace::OTHER;
}
-/// returns true if any instruction in \p MF accesses LDS through DMA.
-static bool containsLDSDMA(const MachineFunction &MF) {
- return any_of(MF, [](const MachineBasicBlock &MBB) {
- return any_of(MBB.instrs(), [](const MachineInstr &MI) {
- return SIInstrInfo::isLDSDMA(MI);
- });
- });
-}
-
// TODO: Consider moving single-wave workgroup->wavefront scope relaxation to an
// IR pass (and extending it to other scoped operations), so middle-end
// optimizations see wavefront scope earlier.
SIMemOpAccess::SIMemOpAccess(const AMDGPUMachineModuleInfo &MMI_,
const GCNSubtarget &ST, const MachineFunction &MF)
- : MMI(&MMI_), ST(ST), CanDemoteWorkgroupToWavefront(
- ST.isSingleWavefrontWorkgroup(MF.getFunction()) &&
- !containsLDSDMA(MF)) {}
+ : MMI(&MMI_), ST(ST),
+ CanDemoteWorkgroupToWavefront(
+ ST.isSingleWavefrontWorkgroup(MF.getFunction()) &&
+ MF.getFunction().hasFnAttribute("amdgpu-no-async")) {}
std::optional<SIMemOpInfo> SIMemOpAccess::constructFromMIWithMMO(
const MachineBasicBlock::iterator &MI) const {
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 3bc50b5017ecfd..decda74eca153a 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -3402,6 +3402,45 @@ bool isIntrinsicAlwaysUniform(unsigned IntrID) {
return lookupAlwaysUniform(IntrID);
}
+bool isAsyncIntrinsic(unsigned IntrID) {
+ switch (IntrID) {
+ case Intrinsic::amdgcn_raw_buffer_load_lds:
+ case Intrinsic::amdgcn_raw_buffer_load_async_lds:
+ case Intrinsic::amdgcn_raw_ptr_buffer_load_lds:
+ case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds:
+ case Intrinsic::amdgcn_struct_buffer_load_lds:
+ case Intrinsic::amdgcn_struct_buffer_load_async_lds:
+ case Intrinsic::amdgcn_struct_ptr_buffer_load_lds:
+ case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds:
+ case Intrinsic::amdgcn_load_to_lds:
+ case Intrinsic::amdgcn_load_async_to_lds:
+ case Intrinsic::amdgcn_global_load_lds:
+ case Intrinsic::amdgcn_global_load_async_lds:
+ case Intrinsic::amdgcn_cluster_load_async_to_lds_b8:
+ case Intrinsic::amdgcn_cluster_load_async_to_lds_b32:
+ case Intrinsic::amdgcn_cluster_load_async_to_lds_b64:
+ case Intrinsic::amdgcn_cluster_load_async_to_lds_b128:
+ case Intrinsic::amdgcn_global_load_async_to_lds_b8:
+ case Intrinsic::amdgcn_global_load_async_to_lds_b32:
+ case Intrinsic::amdgcn_global_load_async_to_lds_b64:
+ case Intrinsic::amdgcn_global_load_async_to_lds_b128:
+ case Intrinsic::amdgcn_global_store_async_from_lds_b8:
+ case Intrinsic::amdgcn_global_store_async_from_lds_b32:
+ case Intrinsic::amdgcn_global_store_async_from_lds_b64:
+ case Intrinsic::amdgcn_global_store_async_from_lds_b128:
+ case Intrinsic::amdgcn_tensor_load_to_lds:
+ case Intrinsic::amdgcn_tensor_store_from_lds:
+ case Intrinsic::amdgcn_asyncmark:
+ case Intrinsic::amdgcn_wait_asyncmark:
+ case Intrinsic::amdgcn_s_wait_asynccnt:
+ case Intrinsic::amdgcn_s_wait_tensorcnt:
+ case Intrinsic::amdgcn_ds_atomic_async_barrier_arrive_b64:
+ return true;
+ default:
+ return false;
+ }
+}
+
const GcnBufferFormatInfo *getGcnBufferFormatInfo(uint8_t BitsPerComp,
uint8_t NumComponents,
uint8_t NumFormat,
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index 309e5e0284b8c3..d8d76dce34b897 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -1762,6 +1762,9 @@ bool isIntrinsicSourceOfDivergence(unsigned IntrID);
/// \returns true if the intrinsic is uniform
bool isIntrinsicAlwaysUniform(unsigned IntrID);
+/// \returns true if the intrinsic executes an asynchronous operation
+bool isAsyncIntrinsic(unsigned IntrID);
+
/// \returns a register class for the physical register \p Reg if it is a VGPR
/// or nullptr otherwise.
const MCRegisterClass *getVGPRPhysRegClass(MCRegister Reg,
diff --git a/llvm/test/CodeGen/AMDGPU/addrspacecast-constantexpr.ll b/llvm/test/CodeGen/AMDGPU/addrspacecast-constantexpr.ll
index 6531cff24af1c4..9534de71c1cb47 100644
--- a/llvm/test/CodeGen/AMDGPU/addrspacecast-constantexpr.ll
+++ b/llvm/test/CodeGen/AMDGPU/addrspacecast-constantexpr.ll
@@ -169,6 +169,6 @@ attributes #1 = { nounwind }
;.
; HSA: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(argmem: readwrite) }
-; HSA: attributes #[[ATTR1]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR2]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR1]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR2]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-flat-scratch-init-asan.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-flat-scratch-init-asan.ll
index 930f1c5abc5cc1..adb6343d98a802 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-flat-scratch-init-asan.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-flat-scratch-init-asan.ll
@@ -20,5 +20,5 @@ define amdgpu_kernel void @k0() #0 {
attributes #0 = { sanitize_address }
; "amdgpu-no-flat-scratch-init" attribute should not be present in attribute list
;.
-; CHECK: attributes #[[ATTR0]] = { sanitize_address "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR0]] = { sanitize_address "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-min-agpr-alloc.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-min-agpr-alloc.ll
index 02eef5ab5e98b8..6bf9e530b228c0 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-min-agpr-alloc.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-min-agpr-alloc.ll
@@ -94,7 +94,7 @@ define amdgpu_kernel void @kernel_uses_asm_virtreg_second_arg() {
define amdgpu_kernel void @kernel_uses_non_agpr_asm() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_non_agpr_asm(
-; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-SAME: ) #[[ATTR3:[0-9]+]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -179,7 +179,7 @@ define amdgpu_kernel void @kernel_calls_extern() {
define amdgpu_kernel void @kernel_calls_extern_marked_callsite() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_calls_extern_marked_callsite() {
-; CHECK-NEXT: call void @unknown() #[[ATTR35:[0-9]+]]
+; CHECK-NEXT: call void @unknown() #[[ATTR37:[0-9]+]]
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
;
@@ -203,7 +203,7 @@ define amdgpu_kernel void @kernel_calls_indirect(ptr %indirect) {
define amdgpu_kernel void @kernel_calls_indirect_marked_callsite(ptr %indirect) {
; CHECK-LABEL: define amdgpu_kernel void @kernel_calls_indirect_marked_callsite(
; CHECK-SAME: ptr [[INDIRECT:%.*]]) {
-; CHECK-NEXT: call void [[INDIRECT]]() #[[ATTR35]]
+; CHECK-NEXT: call void [[INDIRECT]]() #[[ATTR37]]
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
;
@@ -352,7 +352,7 @@ define amdgpu_kernel void @kernel_uses_asm_virtreg_def_struct_0() {
define amdgpu_kernel void @kernel_uses_asm_virtreg_use_struct_1() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_asm_virtreg_use_struct_1(
-; CHECK-SAME: ) #[[ATTR4:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR5:[0-9]+]] {
; CHECK-NEXT: [[DEF:%.*]] = call { i32, <2 x i32> } asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -400,7 +400,7 @@ define amdgpu_kernel void @kernel_uses_asm_virtreg_def_ptr_ty() {
define amdgpu_kernel void @kernel_uses_asm_virtreg_def_vector_ptr_ty() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_asm_virtreg_def_vector_ptr_ty(
-; CHECK-SAME: ) #[[ATTR4]] {
+; CHECK-SAME: ) #[[ATTR5]] {
; CHECK-NEXT: [[DEF:%.*]] = call <2 x ptr> asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -412,7 +412,7 @@ define amdgpu_kernel void @kernel_uses_asm_virtreg_def_vector_ptr_ty() {
define amdgpu_kernel void @kernel_uses_asm_physreg_def_struct_0() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_asm_physreg_def_struct_0(
-; CHECK-SAME: ) #[[ATTR5:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR6:[0-9]+]] {
; CHECK-NEXT: [[DEF:%.*]] = call { i32, i32 } asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -424,7 +424,7 @@ define amdgpu_kernel void @kernel_uses_asm_physreg_def_struct_0() {
define amdgpu_kernel void @kernel_uses_asm_clobber() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_asm_clobber(
-; CHECK-SAME: ) #[[ATTR6:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR7:[0-9]+]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -436,7 +436,7 @@ define amdgpu_kernel void @kernel_uses_asm_clobber() {
define amdgpu_kernel void @kernel_uses_asm_clobber_tuple() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_asm_clobber_tuple(
-; CHECK-SAME: ) #[[ATTR7:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR8:[0-9]+]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -448,7 +448,7 @@ define amdgpu_kernel void @kernel_uses_asm_clobber_tuple() {
define amdgpu_kernel void @kernel_uses_asm_clobber_oob() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_asm_clobber_oob(
-; CHECK-SAME: ) #[[ATTR8:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR9:[0-9]+]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -460,7 +460,7 @@ define amdgpu_kernel void @kernel_uses_asm_clobber_oob() {
define amdgpu_kernel void @kernel_uses_asm_clobber_max() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_asm_clobber_max(
-; CHECK-SAME: ) #[[ATTR8]] {
+; CHECK-SAME: ) #[[ATTR9]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -472,7 +472,7 @@ define amdgpu_kernel void @kernel_uses_asm_clobber_max() {
define amdgpu_kernel void @kernel_uses_asm_physreg_oob() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_asm_physreg_oob(
-; CHECK-SAME: ) #[[ATTR8]] {
+; CHECK-SAME: ) #[[ATTR9]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -484,7 +484,7 @@ define amdgpu_kernel void @kernel_uses_asm_physreg_oob() {
define amdgpu_kernel void @kernel_uses_asm_virtreg_def_max_ty() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_asm_virtreg_def_max_ty(
-; CHECK-SAME: ) #[[ATTR9:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR10:[0-9]+]] {
; CHECK-NEXT: [[DEF:%.*]] = call <32 x i32> asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -496,7 +496,7 @@ define amdgpu_kernel void @kernel_uses_asm_virtreg_def_max_ty() {
define amdgpu_kernel void @kernel_uses_asm_virtreg_use_max_ty() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_asm_virtreg_use_max_ty(
-; CHECK-SAME: ) #[[ATTR9]] {
+; CHECK-SAME: ) #[[ATTR10]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -508,7 +508,7 @@ define amdgpu_kernel void @kernel_uses_asm_virtreg_use_max_ty() {
define amdgpu_kernel void @kernel_uses_asm_virtreg_use_def_max_ty() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_asm_virtreg_use_def_max_ty(
-; CHECK-SAME: ) #[[ATTR9]] {
+; CHECK-SAME: ) #[[ATTR10]] {
; CHECK-NEXT: [[DEF:%.*]] = call <32 x i32> asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -520,7 +520,7 @@ define amdgpu_kernel void @kernel_uses_asm_virtreg_use_def_max_ty() {
define amdgpu_kernel void @vreg_use_exceeds_register_file() {
; CHECK-LABEL: define amdgpu_kernel void @vreg_use_exceeds_register_file(
-; CHECK-SAME: ) #[[ATTR8]] {
+; CHECK-SAME: ) #[[ATTR9]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -532,7 +532,7 @@ define amdgpu_kernel void @vreg_use_exceeds_register_file() {
define amdgpu_kernel void @vreg_def_exceeds_register_file() {
; CHECK-LABEL: define amdgpu_kernel void @vreg_def_exceeds_register_file(
-; CHECK-SAME: ) #[[ATTR8]] {
+; CHECK-SAME: ) #[[ATTR9]] {
; CHECK-NEXT: [[DEF:%.*]] = call <257 x i32> asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -544,7 +544,7 @@ define amdgpu_kernel void @vreg_def_exceeds_register_file() {
define amdgpu_kernel void @multiple() {
; CHECK-LABEL: define amdgpu_kernel void @multiple(
-; CHECK-SAME: ) #[[ATTR9]] {
+; CHECK-SAME: ) #[[ATTR10]] {
; CHECK-NEXT: [[DEF:%.*]] = call { <16 x i32>, <8 x i32>, <8 x i32> } asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -556,7 +556,7 @@ define amdgpu_kernel void @multiple() {
define amdgpu_kernel void @earlyclobber_0() {
; CHECK-LABEL: define amdgpu_kernel void @earlyclobber_0(
-; CHECK-SAME: ) #[[ATTR10:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR11:[0-9]+]] {
; CHECK-NEXT: [[DEF:%.*]] = call <8 x i32> asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -568,7 +568,7 @@ define amdgpu_kernel void @earlyclobber_0() {
define amdgpu_kernel void @earlyclobber_1() {
; CHECK-LABEL: define amdgpu_kernel void @earlyclobber_1(
-; CHECK-SAME: ) #[[ATTR11:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR12:[0-9]+]] {
; CHECK-NEXT: [[DEF:%.*]] = call { <8 x i32>, <16 x i32> } asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -580,7 +580,7 @@ define amdgpu_kernel void @earlyclobber_1() {
define amdgpu_kernel void @physreg_a32__vreg_a256__vreg_a512() {
; CHECK-LABEL: define amdgpu_kernel void @physreg_a32__vreg_a256__vreg_a512(
-; CHECK-SAME: ) #[[ATTR12:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR13:[0-9]+]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -592,7 +592,7 @@ define amdgpu_kernel void @physreg_a32__vreg_a256__vreg_a512() {
define amdgpu_kernel void @physreg_def_a32__def_vreg_a256__def_vreg_a512() {
; CHECK-LABEL: define amdgpu_kernel void @physreg_def_a32__def_vreg_a256__def_vreg_a512(
-; CHECK-SAME: ) #[[ATTR12]] {
+; CHECK-SAME: ) #[[ATTR13]] {
; CHECK-NEXT: [[TMP1:%.*]] = call { i32, <8 x i32>, <16 x i32> } asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -604,7 +604,7 @@ define amdgpu_kernel void @physreg_def_a32__def_vreg_a256__def_vreg_a512() {
define amdgpu_kernel void @physreg_def_a32___def_vreg_a512_use_vreg_a256() {
; CHECK-LABEL: define amdgpu_kernel void @physreg_def_a32___def_vreg_a512_use_vreg_a256(
-; CHECK-SAME: ) #[[ATTR13:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR14:[0-9]+]] {
; CHECK-NEXT: [[TMP1:%.*]] = call { i32, <16 x i32> } asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -616,7 +616,7 @@ define amdgpu_kernel void @physreg_def_a32___def_vreg_a512_use_vreg_a256() {
define amdgpu_kernel void @mixed_physreg_vreg_tuples_0() {
; CHECK-LABEL: define amdgpu_kernel void @mixed_physreg_vreg_tuples_0(
-; CHECK-SAME: ) #[[ATTR10]] {
+; CHECK-SAME: ) #[[ATTR11]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -628,7 +628,7 @@ define amdgpu_kernel void @mixed_physreg_vreg_tuples_0() {
define amdgpu_kernel void @mixed_physreg_vreg_tuples_1() {
; CHECK-LABEL: define amdgpu_kernel void @mixed_physreg_vreg_tuples_1(
-; CHECK-SAME: ) #[[ATTR14:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR15:[0-9]+]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -640,7 +640,7 @@ define amdgpu_kernel void @mixed_physreg_vreg_tuples_1() {
define amdgpu_kernel void @physreg_raises_limit() {
; CHECK-LABEL: define amdgpu_kernel void @physreg_raises_limit(
-; CHECK-SAME: ) #[[ATTR15:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR16:[0-9]+]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -652,7 +652,7 @@ define amdgpu_kernel void @physreg_raises_limit() {
define amdgpu_kernel void @physreg_tuple_alignment_raises_limit() {
; CHECK-LABEL: define amdgpu_kernel void @physreg_tuple_alignment_raises_limit(
-; CHECK-SAME: ) #[[ATTR10]] {
+; CHECK-SAME: ) #[[ATTR11]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -664,7 +664,7 @@ define amdgpu_kernel void @physreg_tuple_alignment_raises_limit() {
define amdgpu_kernel void @align3_virtreg() {
; CHECK-LABEL: define amdgpu_kernel void @align3_virtreg(
-; CHECK-SAME: ) #[[ATTR5]] {
+; CHECK-SAME: ) #[[ATTR6]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -676,7 +676,7 @@ define amdgpu_kernel void @align3_virtreg() {
define amdgpu_kernel void @align3_align4_virtreg() {
; CHECK-LABEL: define amdgpu_kernel void @align3_align4_virtreg(
-; CHECK-SAME: ) #[[ATTR14]] {
+; CHECK-SAME: ) #[[ATTR15]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -688,7 +688,7 @@ define amdgpu_kernel void @align3_align4_virtreg() {
define amdgpu_kernel void @align2_align4_virtreg() {
; CHECK-LABEL: define amdgpu_kernel void @align2_align4_virtreg(
-; CHECK-SAME: ) #[[ATTR14]] {
+; CHECK-SAME: ) #[[ATTR15]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -700,7 +700,7 @@ define amdgpu_kernel void @align2_align4_virtreg() {
define amdgpu_kernel void @kernel_uses_write_register_a55() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_write_register_a55(
-; CHECK-SAME: ) #[[ATTR16:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR17:[0-9]+]] {
; CHECK-NEXT: call void @llvm.write_register.i32(metadata [[META0:![0-9]+]], i32 0)
; CHECK-NEXT: ret void
;
@@ -722,7 +722,7 @@ define amdgpu_kernel void @kernel_uses_write_register_v55() {
define amdgpu_kernel void @kernel_uses_write_register_a55_57() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_write_register_a55_57(
-; CHECK-SAME: ) #[[ATTR17:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR18:[0-9]+]] {
; CHECK-NEXT: call void @llvm.write_register.i96(metadata [[META2:![0-9]+]], i96 0)
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -734,7 +734,7 @@ define amdgpu_kernel void @kernel_uses_write_register_a55_57() {
define amdgpu_kernel void @kernel_uses_read_register_a55(ptr addrspace(1) %ptr) {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_read_register_a55(
-; CHECK-SAME: ptr addrspace(1) [[PTR:%.*]]) #[[ATTR18:[0-9]+]] {
+; CHECK-SAME: ptr addrspace(1) [[PTR:%.*]]) #[[ATTR19:[0-9]+]] {
; CHECK-NEXT: [[REG:%.*]] = call i32 @llvm.read_register.i32(metadata [[META0]])
; CHECK-NEXT: store i32 [[REG]], ptr addrspace(1) [[PTR]], align 4
; CHECK-NEXT: call void @use_most()
@@ -748,7 +748,7 @@ define amdgpu_kernel void @kernel_uses_read_register_a55(ptr addrspace(1) %ptr)
define amdgpu_kernel void @kernel_uses_read_volatile_register_a55(ptr addrspace(1) %ptr) {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_read_volatile_register_a55(
-; CHECK-SAME: ptr addrspace(1) [[PTR:%.*]]) #[[ATTR19:[0-9]+]] {
+; CHECK-SAME: ptr addrspace(1) [[PTR:%.*]]) #[[ATTR20:[0-9]+]] {
; CHECK-NEXT: [[REG:%.*]] = call i32 @llvm.read_volatile_register.i32(metadata [[META0]])
; CHECK-NEXT: store i32 [[REG]], ptr addrspace(1) [[PTR]], align 4
; CHECK-NEXT: call void @use_most()
@@ -762,7 +762,7 @@ define amdgpu_kernel void @kernel_uses_read_volatile_register_a55(ptr addrspace(
define amdgpu_kernel void @kernel_uses_read_register_a56_59(ptr addrspace(1) %ptr) {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_read_register_a56_59(
-; CHECK-SAME: ptr addrspace(1) [[PTR:%.*]]) #[[ATTR20:[0-9]+]] {
+; CHECK-SAME: ptr addrspace(1) [[PTR:%.*]]) #[[ATTR21:[0-9]+]] {
; CHECK-NEXT: [[REG:%.*]] = call i128 @llvm.read_register.i128(metadata [[META3:![0-9]+]])
; CHECK-NEXT: store i128 [[REG]], ptr addrspace(1) [[PTR]], align 8
; CHECK-NEXT: call void @use_most()
@@ -776,7 +776,7 @@ define amdgpu_kernel void @kernel_uses_read_register_a56_59(ptr addrspace(1) %pt
define amdgpu_kernel void @kernel_uses_write_register_out_of_bounds_a256() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_write_register_out_of_bounds_a256(
-; CHECK-SAME: ) #[[ATTR8]] {
+; CHECK-SAME: ) #[[ATTR22:[0-9]+]] {
; CHECK-NEXT: call void @llvm.write_register.i32(metadata [[META4:![0-9]+]], i32 0)
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -788,7 +788,7 @@ define amdgpu_kernel void @kernel_uses_write_register_out_of_bounds_a256() {
define amdgpu_kernel void @kernel_multiple_uses() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_multiple_uses(
-; CHECK-SAME: ) #[[ATTR4]] {
+; CHECK-SAME: ) #[[ATTR5]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void asm sideeffect "
@@ -804,7 +804,7 @@ define amdgpu_kernel void @kernel_multiple_uses() {
define amdgpu_kernel void @kernel_multiple_defs() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_multiple_defs(
-; CHECK-SAME: ) #[[ATTR4]] {
+; CHECK-SAME: ) #[[ATTR5]] {
; CHECK-NEXT: [[TMP1:%.*]] = call i64 asm sideeffect "
; CHECK-NEXT: [[TMP2:%.*]] = call i32 asm sideeffect "
; CHECK-NEXT: [[TMP3:%.*]] = call i128 asm sideeffect "
@@ -820,7 +820,7 @@ define amdgpu_kernel void @kernel_multiple_defs() {
define amdgpu_kernel void @kernel_multiple_use_defs() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_multiple_use_defs(
-; CHECK-SAME: ) #[[ATTR4]] {
+; CHECK-SAME: ) #[[ATTR5]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: [[TMP1:%.*]] = call i128 asm sideeffect "
; CHECK-NEXT: call void @use_most()
@@ -834,7 +834,7 @@ define amdgpu_kernel void @kernel_multiple_use_defs() {
define void @callgraph_b() {
; CHECK-LABEL: define void @callgraph_b(
-; CHECK-SAME: ) #[[ATTR14]] {
+; CHECK-SAME: ) #[[ATTR15]] {
; CHECK-NEXT: [[TMP1:%.*]] = call <4 x i32> asm sideeffect "
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
@@ -862,7 +862,7 @@ define void @callgraph_c() {
define void @callgraph_a(i1 %cond) {
; CHECK-LABEL: define void @callgraph_a(
-; CHECK-SAME: i1 [[COND:%.*]]) #[[ATTR14]] {
+; CHECK-SAME: i1 [[COND:%.*]]) #[[ATTR15]] {
; CHECK-NEXT: br i1 [[COND]], label [[A:%.*]], label [[B:%.*]]
; CHECK: a:
; CHECK-NEXT: call void @callgraph_b()
@@ -885,7 +885,7 @@ b:
define void @kernel_max_callgraph(i1 %cond) {
; CHECK-LABEL: define void @kernel_max_callgraph(
-; CHECK-SAME: i1 [[COND:%.*]]) #[[ATTR14]] {
+; CHECK-SAME: i1 [[COND:%.*]]) #[[ATTR15]] {
; CHECK-NEXT: call void @callgraph_a(i1 [[COND]])
; CHECK-NEXT: ret void
;
@@ -895,7 +895,7 @@ define void @kernel_max_callgraph(i1 %cond) {
define amdgpu_kernel void @kernel_uses_all_virtregs() #1 {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_all_virtregs(
-; CHECK-SAME: ) #[[ATTR21:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR23:[0-9]+]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -907,7 +907,7 @@ define amdgpu_kernel void @kernel_uses_all_virtregs() #1 {
define amdgpu_kernel void @kernel_uses_all_virtregs_plus_1() #1 {
; CHECK-LABEL: define amdgpu_kernel void @kernel_uses_all_virtregs_plus_1(
-; CHECK-SAME: ) #[[ATTR21]] {
+; CHECK-SAME: ) #[[ATTR23]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -919,7 +919,7 @@ define amdgpu_kernel void @kernel_uses_all_virtregs_plus_1() #1 {
define void @recursive() {
; CHECK-LABEL: define void @recursive(
-; CHECK-SAME: ) #[[ATTR22:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR24:[0-9]+]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: call void @recursive()
@@ -933,7 +933,7 @@ define void @recursive() {
define void @indirect_0() {
; CHECK-LABEL: define void @indirect_0(
-; CHECK-SAME: ) #[[ATTR22]] {
+; CHECK-SAME: ) #[[ATTR24]] {
; CHECK-NEXT: call void asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -945,7 +945,7 @@ define void @indirect_0() {
define void @indirect_1() {
; CHECK-LABEL: define void @indirect_1(
-; CHECK-SAME: ) #[[ATTR23:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR25:[0-9]+]] {
; CHECK-NEXT: [[TMP1:%.*]] = call <3 x i32> asm sideeffect "
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -957,7 +957,7 @@ define void @indirect_1() {
define amdgpu_kernel void @knowable_indirect_call(i1 %cond) {
; CHECK-LABEL: define amdgpu_kernel void @knowable_indirect_call(
-; CHECK-SAME: i1 [[COND:%.*]]) #[[ATTR22]] {
+; CHECK-SAME: i1 [[COND:%.*]]) #[[ATTR24]] {
; CHECK-NEXT: [[FPTR:%.*]] = select i1 [[COND]], ptr @indirect_0, ptr @indirect_1
; CHECK-NEXT: [[TMP1:%.*]] = icmp eq ptr [[FPTR]], @indirect_1
; CHECK-NEXT: br i1 [[TMP1]], label [[TMP2:%.*]], label [[TMP3:%.*]]
@@ -1021,7 +1021,7 @@ define amdgpu_kernel void @indirect_unknown(ptr %fptr) {
define amdgpu_kernel void @kernel_sanitize_address() sanitize_address {
; CHECK: Function Attrs: sanitize_address
; CHECK-LABEL: define amdgpu_kernel void @kernel_sanitize_address(
-; CHECK-SAME: ) #[[ATTR24:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR26:[0-9]+]] {
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
;
@@ -1032,7 +1032,7 @@ define amdgpu_kernel void @kernel_sanitize_address() sanitize_address {
define amdgpu_kernel void @kernel_sanitize_memory() sanitize_memory {
; CHECK: Function Attrs: sanitize_memory
; CHECK-LABEL: define amdgpu_kernel void @kernel_sanitize_memory(
-; CHECK-SAME: ) #[[ATTR25:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR27:[0-9]+]] {
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
;
@@ -1043,7 +1043,7 @@ define amdgpu_kernel void @kernel_sanitize_memory() sanitize_memory {
define amdgpu_kernel void @kernel_sanitize_thread() sanitize_thread {
; CHECK: Function Attrs: sanitize_thread
; CHECK-LABEL: define amdgpu_kernel void @kernel_sanitize_thread(
-; CHECK-SAME: ) #[[ATTR26:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR28:[0-9]+]] {
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
;
@@ -1054,7 +1054,7 @@ define amdgpu_kernel void @kernel_sanitize_thread() sanitize_thread {
define amdgpu_kernel void @kernel_sanitize_hwaddress() sanitize_hwaddress {
; CHECK: Function Attrs: sanitize_hwaddress
; CHECK-LABEL: define amdgpu_kernel void @kernel_sanitize_hwaddress(
-; CHECK-SAME: ) #[[ATTR27:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR29:[0-9]+]] {
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
;
@@ -1069,7 +1069,7 @@ define amdgpu_kernel void @kernel_sanitize_hwaddress() sanitize_hwaddress {
define amdgpu_kernel void @kernel_sanitize_address_preannotated() #2 {
; CHECK: Function Attrs: sanitize_address
; CHECK-LABEL: define amdgpu_kernel void @kernel_sanitize_address_preannotated(
-; CHECK-SAME: ) #[[ATTR28:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR30:[0-9]+]] {
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
;
@@ -1082,7 +1082,7 @@ define amdgpu_kernel void @kernel_sanitize_address_preannotated() #2 {
define void @sanitized_callee() sanitize_address {
; CHECK: Function Attrs: sanitize_address
; CHECK-LABEL: define void @sanitized_callee(
-; CHECK-SAME: ) #[[ATTR24]] {
+; CHECK-SAME: ) #[[ATTR26]] {
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
;
@@ -1092,7 +1092,7 @@ define void @sanitized_callee() sanitize_address {
define amdgpu_kernel void @kernel_calls_sanitized_callee() {
; CHECK-LABEL: define amdgpu_kernel void @kernel_calls_sanitized_callee(
-; CHECK-SAME: ) #[[ATTR29:[0-9]+]] {
+; CHECK-SAME: ) #[[ATTR31:[0-9]+]] {
; CHECK-NEXT: call void @sanitized_callee()
; CHECK-NEXT: call void @use_most()
; CHECK-NEXT: ret void
@@ -1113,42 +1113,44 @@ attributes #2 = { sanitize_address "amdgpu-agpr-alloc"="0" }
!4 = !{!"a256"}
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-agpr-alloc"="0" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-agpr-alloc"="0" "amdgpu-no-async" "amdgpu-no-wwm" }
; CHECK: attributes #[[ATTR1]] = { "amdgpu-agpr-alloc"="1" "amdgpu-no-wwm" }
; CHECK: attributes #[[ATTR2]] = { "amdgpu-agpr-alloc"="2" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR3:[0-9]+]] = { convergent nocallback nocreateundeforpoison nofree nosync nounwind willreturn memory(none) }
-; CHECK: attributes #[[ATTR4]] = { "amdgpu-agpr-alloc"="4" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR5]] = { "amdgpu-agpr-alloc"="6" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR6]] = { "amdgpu-agpr-alloc"="5" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR7]] = { "amdgpu-agpr-alloc"="14" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR8]] = { "amdgpu-agpr-alloc"="256" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR9]] = { "amdgpu-agpr-alloc"="32" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR10]] = { "amdgpu-agpr-alloc"="9" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR11]] = { "amdgpu-agpr-alloc"="64" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR12]] = { "amdgpu-agpr-alloc"="49" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR13]] = { "amdgpu-agpr-alloc"="33" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR14]] = { "amdgpu-agpr-alloc"="8" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR15]] = { "amdgpu-agpr-alloc"="13" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR16]] = { "amdgpu-agpr-alloc"="56" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR17]] = { "amdgpu-agpr-alloc"="58" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR18]] = { "amdgpu-agpr-alloc"="56" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR19]] = { "amdgpu-agpr-alloc"="56" }
-; CHECK: attributes #[[ATTR20]] = { "amdgpu-agpr-alloc"="60" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR21]] = { "amdgpu-agpr-alloc"="256" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="1,1" }
-; CHECK: attributes #[[ATTR22]] = { "amdgpu-agpr-alloc"="7" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR23]] = { "amdgpu-agpr-alloc"="3" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR24]] = { sanitize_address "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR25]] = { sanitize_memory "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR26]] = { sanitize_thread "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR27]] = { sanitize_hwaddress "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR28]] = { sanitize_address "amdgpu-agpr-alloc"="0" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR29]] = { "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR30:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
-; CHECK: attributes #[[ATTR31:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(argmem: readwrite) }
-; CHECK: attributes #[[ATTR32:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(read) }
-; CHECK: attributes #[[ATTR33:[0-9]+]] = { nounwind }
-; CHECK: attributes #[[ATTR34:[0-9]+]] = { nocallback nounwind }
-; CHECK: attributes #[[ATTR35]] = { "amdgpu-agpr-alloc"="0" }
+; CHECK: attributes #[[ATTR3]] = { "amdgpu-agpr-alloc"="0" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR4:[0-9]+]] = { convergent nocallback nocreateundeforpoison nofree nosync nounwind willreturn memory(none) }
+; CHECK: attributes #[[ATTR5]] = { "amdgpu-agpr-alloc"="4" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR6]] = { "amdgpu-agpr-alloc"="6" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR7]] = { "amdgpu-agpr-alloc"="5" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR8]] = { "amdgpu-agpr-alloc"="14" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR9]] = { "amdgpu-agpr-alloc"="256" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR10]] = { "amdgpu-agpr-alloc"="32" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR11]] = { "amdgpu-agpr-alloc"="9" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR12]] = { "amdgpu-agpr-alloc"="64" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR13]] = { "amdgpu-agpr-alloc"="49" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR14]] = { "amdgpu-agpr-alloc"="33" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR15]] = { "amdgpu-agpr-alloc"="8" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR16]] = { "amdgpu-agpr-alloc"="13" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR17]] = { "amdgpu-agpr-alloc"="56" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR18]] = { "amdgpu-agpr-alloc"="58" "amdgpu-no-async" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR19]] = { "amdgpu-agpr-alloc"="56" "amdgpu-no-async" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR20]] = { "amdgpu-agpr-alloc"="56" }
+; CHECK: attributes #[[ATTR21]] = { "amdgpu-agpr-alloc"="60" "amdgpu-no-async" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR22]] = { "amdgpu-agpr-alloc"="256" "amdgpu-no-async" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR23]] = { "amdgpu-agpr-alloc"="256" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="1,1" }
+; CHECK: attributes #[[ATTR24]] = { "amdgpu-agpr-alloc"="7" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR25]] = { "amdgpu-agpr-alloc"="3" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR26]] = { sanitize_address "amdgpu-no-async" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR27]] = { sanitize_memory "amdgpu-no-async" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR28]] = { sanitize_thread "amdgpu-no-async" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR29]] = { sanitize_hwaddress "amdgpu-no-async" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR30]] = { sanitize_address "amdgpu-agpr-alloc"="0" "amdgpu-no-async" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR31]] = { "amdgpu-no-async" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR32:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
+; CHECK: attributes #[[ATTR33:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(argmem: readwrite) }
+; CHECK: attributes #[[ATTR34:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(read) }
+; CHECK: attributes #[[ATTR35:[0-9]+]] = { nounwind }
+; CHECK: attributes #[[ATTR36:[0-9]+]] = { nocallback nounwind }
+; CHECK: attributes #[[ATTR37]] = { "amdgpu-agpr-alloc"="0" }
;.
; CHECK: [[META0]] = !{!"a55"}
; CHECK: [[META1]] = !{!"v55"}
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-no-async.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-no-async.ll
new file mode 100644
index 00000000000000..2fad15f6fd3430
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-no-async.ll
@@ -0,0 +1,286 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-attributes --check-globals all --version 6
+; RUN: opt -S -mtriple=amdgpu12.50-amd-amdhsa -passes=amdgpu-attributor %s | FileCheck %s
+
+declare void @llvm.amdgcn.raw.ptr.buffer.load.lds(ptr addrspace(8), ptr addrspace(3), i32, i32, i32, i32, i32)
+declare void @llvm.amdgcn.load.to.lds.p1(ptr addrspace(1), ptr addrspace(3), i32, i32, i32)
+declare void @llvm.amdgcn.global.load.async.to.lds.b32(ptr addrspace(1), ptr addrspace(3), i32, i32)
+declare void @llvm.amdgcn.global.store.async.from.lds.b32(ptr addrspace(1), ptr addrspace(3), i32, i32)
+declare void @llvm.amdgcn.tensor.load.to.lds(<4 x i32>, <8 x i32>, <4 x i32>, <4 x i32>, <8 x i32>, i32)
+declare void @llvm.amdgcn.tensor.store.from.lds(<4 x i32>, <8 x i32>, <4 x i32>, <4 x i32>, <8 x i32>, i32)
+declare void @llvm.amdgcn.asyncmark()
+declare void @llvm.amdgcn.wait.asyncmark(i16 immarg)
+declare void @llvm.amdgcn.s.wait.asynccnt(i16 immarg)
+declare void @llvm.amdgcn.s.wait.tensorcnt(i16 immarg)
+declare void @llvm.amdgcn.ds.atomic.async.barrier.arrive.b64(ptr addrspace(3))
+declare void @llvm.amdgcn.s.barrier()
+
+declare void @unknown_extern()
+declare void @known_clean_extern() #0
+
+define amdgpu_kernel void @no_async(ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @no_async(
+; CHECK-SAME: ptr addrspace(3) [[LDS:%.*]]) #[[ATTR7:[0-9]+]] {
+; CHECK-NEXT: store i32 0, ptr addrspace(3) [[LDS]], align 4
+; CHECK-NEXT: ret void
+;
+ store i32 0, ptr addrspace(3) %lds
+ ret void
+}
+
+define amdgpu_kernel void @asyncmark(ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @asyncmark(
+; CHECK-SAME: ptr addrspace(3) [[LDS:%.*]]) #[[ATTR8:[0-9]+]] {
+; CHECK-NEXT: call void @llvm.amdgcn.asyncmark()
+; CHECK-NEXT: call void @llvm.amdgcn.wait.asyncmark(i16 0)
+; CHECK-NEXT: store i32 0, ptr addrspace(3) [[LDS]], align 4
+; CHECK-NEXT: ret void
+;
+ call void @llvm.amdgcn.asyncmark()
+ call void @llvm.amdgcn.wait.asyncmark(i16 0)
+ store i32 0, ptr addrspace(3) %lds
+ ret void
+}
+
+define amdgpu_kernel void @wait_async_counters(ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @wait_async_counters(
+; CHECK-SAME: ptr addrspace(3) [[LDS:%.*]]) #[[ATTR8]] {
+; CHECK-NEXT: call void @llvm.amdgcn.s.wait.asynccnt(i16 0)
+; CHECK-NEXT: call void @llvm.amdgcn.s.wait.tensorcnt(i16 0)
+; CHECK-NEXT: store i32 0, ptr addrspace(3) [[LDS]], align 4
+; CHECK-NEXT: ret void
+;
+ call void @llvm.amdgcn.s.wait.asynccnt(i16 0)
+ call void @llvm.amdgcn.s.wait.tensorcnt(i16 0)
+ store i32 0, ptr addrspace(3) %lds
+ ret void
+}
+
+define amdgpu_kernel void @async_barrier_arrive(ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @async_barrier_arrive(
+; CHECK-SAME: ptr addrspace(3) [[LDS:%.*]]) #[[ATTR8]] {
+; CHECK-NEXT: call void @llvm.amdgcn.ds.atomic.async.barrier.arrive.b64(ptr addrspace(3) [[LDS]])
+; CHECK-NEXT: ret void
+;
+ call void @llvm.amdgcn.ds.atomic.async.barrier.arrive.b64(ptr addrspace(3) %lds)
+ ret void
+}
+
+define amdgpu_kernel void @barrier(ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @barrier(
+; CHECK-SAME: ptr addrspace(3) [[LDS:%.*]]) #[[ATTR7]] {
+; CHECK-NEXT: call void @llvm.amdgcn.s.barrier()
+; CHECK-NEXT: store i32 0, ptr addrspace(3) [[LDS]], align 4
+; CHECK-NEXT: ret void
+;
+ call void @llvm.amdgcn.s.barrier()
+ store i32 0, ptr addrspace(3) %lds
+ ret void
+}
+
+define amdgpu_kernel void @buffer_load_lds(ptr addrspace(8) %rsrc, ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @buffer_load_lds(
+; CHECK-SAME: ptr addrspace(8) [[RSRC:%.*]], ptr addrspace(3) [[LDS:%.*]]) #[[ATTR8]] {
+; CHECK-NEXT: call void @llvm.amdgcn.raw.ptr.buffer.load.lds(ptr addrspace(8) [[RSRC]], ptr addrspace(3) [[LDS]], i32 4, i32 0, i32 0, i32 0, i32 0)
+; CHECK-NEXT: ret void
+;
+ call void @llvm.amdgcn.raw.ptr.buffer.load.lds(ptr addrspace(8) %rsrc, ptr addrspace(3) %lds, i32 4, i32 0, i32 0, i32 0, i32 0)
+ ret void
+}
+
+define amdgpu_kernel void @load_to_lds(ptr addrspace(1) %g, ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @load_to_lds(
+; CHECK-SAME: ptr addrspace(1) [[G:%.*]], ptr addrspace(3) [[LDS:%.*]]) #[[ATTR8]] {
+; CHECK-NEXT: call void @llvm.amdgcn.load.to.lds.p1(ptr addrspace(1) [[G]], ptr addrspace(3) [[LDS]], i32 4, i32 0, i32 0)
+; CHECK-NEXT: ret void
+;
+ call void @llvm.amdgcn.load.to.lds.p1(ptr addrspace(1) %g, ptr addrspace(3) %lds, i32 4, i32 0, i32 0)
+ ret void
+}
+
+define amdgpu_kernel void @global_load_async_to_lds(ptr addrspace(1) %g, ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @global_load_async_to_lds(
+; CHECK-SAME: ptr addrspace(1) [[G:%.*]], ptr addrspace(3) [[LDS:%.*]]) #[[ATTR8]] {
+; CHECK-NEXT: call void @llvm.amdgcn.global.load.async.to.lds.b32(ptr addrspace(1) [[G]], ptr addrspace(3) [[LDS]], i32 0, i32 0)
+; CHECK-NEXT: ret void
+;
+ call void @llvm.amdgcn.global.load.async.to.lds.b32(ptr addrspace(1) %g, ptr addrspace(3) %lds, i32 0, i32 0)
+ ret void
+}
+
+define amdgpu_kernel void @global_store_async_from_lds(ptr addrspace(1) %g, ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @global_store_async_from_lds(
+; CHECK-SAME: ptr addrspace(1) [[G:%.*]], ptr addrspace(3) [[LDS:%.*]]) #[[ATTR8]] {
+; CHECK-NEXT: call void @llvm.amdgcn.global.store.async.from.lds.b32(ptr addrspace(1) [[G]], ptr addrspace(3) [[LDS]], i32 0, i32 0)
+; CHECK-NEXT: ret void
+;
+ call void @llvm.amdgcn.global.store.async.from.lds.b32(ptr addrspace(1) %g, ptr addrspace(3) %lds, i32 0, i32 0)
+ ret void
+}
+
+define amdgpu_kernel void @tensor_load_to_lds() {
+; CHECK-LABEL: define amdgpu_kernel void @tensor_load_to_lds(
+; CHECK-SAME: ) #[[ATTR8]] {
+; CHECK-NEXT: call void @llvm.amdgcn.tensor.load.to.lds(<4 x i32> zeroinitializer, <8 x i32> zeroinitializer, <4 x i32> zeroinitializer, <4 x i32> zeroinitializer, <8 x i32> zeroinitializer, i32 0)
+; CHECK-NEXT: ret void
+;
+ call void @llvm.amdgcn.tensor.load.to.lds(<4 x i32> zeroinitializer, <8 x i32> zeroinitializer, <4 x i32> zeroinitializer, <4 x i32> zeroinitializer, <8 x i32> zeroinitializer, i32 0)
+ ret void
+}
+
+define amdgpu_kernel void @tensor_store_from_lds() {
+; CHECK-LABEL: define amdgpu_kernel void @tensor_store_from_lds(
+; CHECK-SAME: ) #[[ATTR8]] {
+; CHECK-NEXT: call void @llvm.amdgcn.tensor.store.from.lds(<4 x i32> zeroinitializer, <8 x i32> zeroinitializer, <4 x i32> zeroinitializer, <4 x i32> zeroinitializer, <8 x i32> zeroinitializer, i32 0)
+; CHECK-NEXT: ret void
+;
+ call void @llvm.amdgcn.tensor.store.from.lds(<4 x i32> zeroinitializer, <8 x i32> zeroinitializer, <4 x i32> zeroinitializer, <4 x i32> zeroinitializer, <8 x i32> zeroinitializer, i32 0)
+ ret void
+}
+
+define internal void @callee_no_async(ptr addrspace(3) %lds) {
+; CHECK-LABEL: define internal void @callee_no_async(
+; CHECK-SAME: ptr addrspace(3) [[LDS:%.*]]) #[[ATTR7]] {
+; CHECK-NEXT: store i32 0, ptr addrspace(3) [[LDS]], align 4
+; CHECK-NEXT: ret void
+;
+ store i32 0, ptr addrspace(3) %lds
+ ret void
+}
+
+define internal void @callee_async(ptr addrspace(1) %g, ptr addrspace(3) %lds) {
+; CHECK-LABEL: define internal void @callee_async(
+; CHECK-SAME: ptr addrspace(1) [[G:%.*]], ptr addrspace(3) [[LDS:%.*]]) #[[ATTR8]] {
+; CHECK-NEXT: call void @llvm.amdgcn.load.to.lds.p1(ptr addrspace(1) [[G]], ptr addrspace(3) [[LDS]], i32 4, i32 0, i32 0)
+; CHECK-NEXT: ret void
+;
+ call void @llvm.amdgcn.load.to.lds.p1(ptr addrspace(1) %g, ptr addrspace(3) %lds, i32 4, i32 0, i32 0)
+ ret void
+}
+
+define internal void @callee_calls_async(ptr addrspace(1) %g, ptr addrspace(3) %lds) {
+; CHECK-LABEL: define internal void @callee_calls_async(
+; CHECK-SAME: ptr addrspace(1) [[G:%.*]], ptr addrspace(3) [[LDS:%.*]]) #[[ATTR8]] {
+; CHECK-NEXT: call void @callee_async(ptr addrspace(1) [[G]], ptr addrspace(3) [[LDS]])
+; CHECK-NEXT: ret void
+;
+ call void @callee_async(ptr addrspace(1) %g, ptr addrspace(3) %lds)
+ ret void
+}
+
+define amdgpu_kernel void @call_no_async(ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @call_no_async(
+; CHECK-SAME: ptr addrspace(3) [[LDS:%.*]]) #[[ATTR7]] {
+; CHECK-NEXT: call void @callee_no_async(ptr addrspace(3) [[LDS]])
+; CHECK-NEXT: ret void
+;
+ call void @callee_no_async(ptr addrspace(3) %lds)
+ ret void
+}
+
+define amdgpu_kernel void @call_async(ptr addrspace(1) %g, ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @call_async(
+; CHECK-SAME: ptr addrspace(1) [[G:%.*]], ptr addrspace(3) [[LDS:%.*]]) #[[ATTR8]] {
+; CHECK-NEXT: call void @callee_async(ptr addrspace(1) [[G]], ptr addrspace(3) [[LDS]])
+; CHECK-NEXT: ret void
+;
+ call void @callee_async(ptr addrspace(1) %g, ptr addrspace(3) %lds)
+ ret void
+}
+
+define amdgpu_kernel void @call_calls_async(ptr addrspace(1) %g, ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @call_calls_async(
+; CHECK-SAME: ptr addrspace(1) [[G:%.*]], ptr addrspace(3) [[LDS:%.*]]) #[[ATTR8]] {
+; CHECK-NEXT: call void @callee_calls_async(ptr addrspace(1) [[G]], ptr addrspace(3) [[LDS]])
+; CHECK-NEXT: ret void
+;
+ call void @callee_calls_async(ptr addrspace(1) %g, ptr addrspace(3) %lds)
+ ret void
+}
+
+define internal void @recurse_no_async(i32 %n, ptr addrspace(3) %lds) {
+; CHECK-LABEL: define internal void @recurse_no_async(
+; CHECK-SAME: i32 [[N:%.*]], ptr addrspace(3) [[LDS:%.*]]) #[[ATTR7]] {
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i32 [[N]], 0
+; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[RECURSE:.*]]
+; CHECK: [[RECURSE]]:
+; CHECK-NEXT: [[DEC:%.*]] = sub i32 [[N]], 1
+; CHECK-NEXT: call void @recurse_no_async(i32 [[DEC]], ptr addrspace(3) [[LDS]])
+; CHECK-NEXT: br label %[[EXIT]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: store i32 0, ptr addrspace(3) [[LDS]], align 4
+; CHECK-NEXT: ret void
+;
+ %cmp = icmp eq i32 %n, 0
+ br i1 %cmp, label %exit, label %recurse
+
+recurse:
+ %dec = sub i32 %n, 1
+ call void @recurse_no_async(i32 %dec, ptr addrspace(3) %lds)
+ br label %exit
+
+exit:
+ store i32 0, ptr addrspace(3) %lds
+ ret void
+}
+
+define amdgpu_kernel void @call_recurse_no_async(i32 %n, ptr addrspace(3) %lds) {
+; CHECK-LABEL: define amdgpu_kernel void @call_recurse_no_async(
+; CHECK-SAME: i32 [[N:%.*]], ptr addrspace(3) [[LDS:%.*]]) #[[ATTR7]] {
+; CHECK-NEXT: call void @recurse_no_async(i32 [[N]], ptr addrspace(3) [[LDS]])
+; CHECK-NEXT: ret void
+;
+ call void @recurse_no_async(i32 %n, ptr addrspace(3) %lds)
+ ret void
+}
+
+define amdgpu_kernel void @call_unknown_extern() {
+; CHECK-LABEL: define amdgpu_kernel void @call_unknown_extern() {
+; CHECK-NEXT: call void @unknown_extern()
+; CHECK-NEXT: ret void
+;
+ call void @unknown_extern()
+ ret void
+}
+
+define amdgpu_kernel void @call_known_clean_extern() {
+; CHECK-LABEL: define amdgpu_kernel void @call_known_clean_extern(
+; CHECK-SAME: ) #[[ATTR6:[0-9]+]] {
+; CHECK-NEXT: call void @known_clean_extern()
+; CHECK-NEXT: ret void
+;
+ call void @known_clean_extern()
+ ret void
+}
+
+define amdgpu_kernel void @call_indirect(ptr %fptr) {
+; CHECK-LABEL: define amdgpu_kernel void @call_indirect(
+; CHECK-SAME: ptr [[FPTR:%.*]]) {
+; CHECK-NEXT: call void [[FPTR]]()
+; CHECK-NEXT: ret void
+;
+ call void %fptr()
+ ret void
+}
+
+define amdgpu_kernel void @inline_asm() {
+; CHECK-LABEL: define amdgpu_kernel void @inline_asm(
+; CHECK-SAME: ) #[[ATTR8]] {
+; CHECK-NEXT: call void asm sideeffect "s_nop 0", ""()
+; CHECK-NEXT: ret void
+;
+ call void asm sideeffect "s_nop 0", ""()
+ ret void
+}
+
+attributes #0 = { "amdgpu-no-async" }
+;.
+; CHECK: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nounwind willreturn memory(argmem: readwrite) }
+; CHECK: attributes #[[ATTR1:[0-9]+]] = { nocallback nofree nounwind willreturn memory(argmem: readwrite, inaccessiblemem: readwrite) }
+; CHECK: attributes #[[ATTR2:[0-9]+]] = { convergent nocallback nofree nounwind willreturn memory(argmem: readwrite, inaccessiblemem: readwrite) }
+; CHECK: attributes #[[ATTR3:[0-9]+]] = { nocallback nofree nounwind willreturn }
+; CHECK: attributes #[[ATTR4:[0-9]+]] = { nocallback nofree nounwind willreturn memory(inaccessiblemem: readwrite) }
+; CHECK: attributes #[[ATTR5:[0-9]+]] = { convergent nocallback nofree nounwind willreturn }
+; CHECK: attributes #[[ATTR6]] = { "amdgpu-no-async" }
+; CHECK: attributes #[[ATTR7]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR8]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+;.
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-nocallback-intrinsics.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-nocallback-intrinsics.ll
index 286df25d5fe23f..065ac4e365850b 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-nocallback-intrinsics.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-nocallback-intrinsics.ll
@@ -35,7 +35,7 @@ define void @use_assume(i1 %arg) {
define void @use_trap() {
; CHECK-LABEL: define void @use_trap(
-; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-SAME: ) #[[ATTR1:[0-9]+]] {
; CHECK-NEXT: call void @llvm.trap()
; CHECK-NEXT: ret void
;
@@ -45,8 +45,8 @@ define void @use_trap() {
define void @use_trap_with_handler() {
; CHECK-LABEL: define void @use_trap_with_handler(
-; CHECK-SAME: ) #[[ATTR1:[0-9]+]] {
-; CHECK-NEXT: call void @llvm.trap() #[[ATTR7:[0-9]+]]
+; CHECK-SAME: ) #[[ATTR2:[0-9]+]] {
+; CHECK-NEXT: call void @llvm.trap() #[[ATTR8:[0-9]+]]
; CHECK-NEXT: ret void
;
call void @llvm.trap() #0
@@ -55,7 +55,7 @@ define void @use_trap_with_handler() {
define void @use_debugtrap() {
; CHECK-LABEL: define void @use_debugtrap(
-; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-SAME: ) #[[ATTR1]] {
; CHECK-NEXT: call void @llvm.debugtrap()
; CHECK-NEXT: ret void
;
@@ -65,7 +65,7 @@ define void @use_debugtrap() {
define void @use_ubsantrap() {
; CHECK-LABEL: define void @use_ubsantrap(
-; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-SAME: ) #[[ATTR1]] {
; CHECK-NEXT: call void @llvm.ubsantrap(i8 0)
; CHECK-NEXT: ret void
;
@@ -76,12 +76,13 @@ define void @use_ubsantrap() {
attributes #0 = { "trap-func-name"="handler" }
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-agpr-alloc"="0" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR2:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(inaccessiblemem: write) }
-; CHECK: attributes #[[ATTR3:[0-9]+]] = { nounwind }
-; CHECK: attributes #[[ATTR4:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(none) }
-; CHECK: attributes #[[ATTR5:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(inaccessiblemem: readwrite) }
-; CHECK: attributes #[[ATTR6:[0-9]+]] = { cold noreturn nounwind memory(inaccessiblemem: write) }
-; CHECK: attributes #[[ATTR7]] = { "trap-func-name"="handler" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-agpr-alloc"="0" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR1]] = { "amdgpu-agpr-alloc"="0" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR2]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR3:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(inaccessiblemem: write) }
+; CHECK: attributes #[[ATTR4:[0-9]+]] = { nounwind }
+; CHECK: attributes #[[ATTR5:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(none) }
+; CHECK: attributes #[[ATTR6:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(inaccessiblemem: readwrite) }
+; CHECK: attributes #[[ATTR7:[0-9]+]] = { cold noreturn nounwind memory(inaccessiblemem: write) }
+; CHECK: attributes #[[ATTR8]] = { "trap-func-name"="handler" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-trap-leaf.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-trap-leaf.ll
index 31d196655bc142..5af0401601de08 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-trap-leaf.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-attributor-trap-leaf.ll
@@ -24,7 +24,7 @@ define amdgpu_kernel void @trap_kernel() {
define amdgpu_kernel void @trap_kernel_with_handler() {
; CHECK-LABEL: define amdgpu_kernel void @trap_kernel_with_handler(
; CHECK-SAME: ) #[[ATTR3:[0-9]+]] {
-; CHECK-NEXT: call void @llvm.trap() #[[ATTR4:[0-9]+]]
+; CHECK-NEXT: call void @llvm.trap() #[[ATTR5:[0-9]+]]
; CHECK-NEXT: ret void
;
call void @llvm.trap() #0
@@ -44,8 +44,8 @@ define amdgpu_kernel void @debugtrap_kernel() {
; Test that a trap with both trap-func-name and nocallback is still safe
define amdgpu_kernel void @trap_kernel_with_handler_and_nocallback() {
; CHECK-LABEL: define amdgpu_kernel void @trap_kernel_with_handler_and_nocallback(
-; CHECK-SAME: ) #[[ATTR2]] {
-; CHECK-NEXT: call void @llvm.trap() #[[ATTR5:[0-9]+]]
+; CHECK-SAME: ) #[[ATTR4:[0-9]+]] {
+; CHECK-NEXT: call void @llvm.trap() #[[ATTR6:[0-9]+]]
; CHECK-NEXT: ret void
;
call void @llvm.trap() #1
@@ -60,6 +60,7 @@ attributes #1 = { nocallback "trap-func-name"="handler" }
; CHECK: attributes #[[ATTR1:[0-9]+]] = { nounwind }
; CHECK: attributes #[[ATTR2]] = { "amdgpu-agpr-alloc"="0" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
; CHECK: attributes #[[ATTR3]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR4]] = { "trap-func-name"="handler" }
-; CHECK: attributes #[[ATTR5]] = { nocallback "trap-func-name"="handler" }
+; CHECK: attributes #[[ATTR4]] = { "amdgpu-agpr-alloc"="0" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR5]] = { "trap-func-name"="handler" }
+; CHECK: attributes #[[ATTR6]] = { nocallback "trap-func-name"="handler" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/annotate-kernel-features-hsa-call-addrspacecast.ll b/llvm/test/CodeGen/AMDGPU/annotate-kernel-features-hsa-call-addrspacecast.ll
index 3b96f2cfe7a2db..1c14af0f83aa92 100644
--- a/llvm/test/CodeGen/AMDGPU/annotate-kernel-features-hsa-call-addrspacecast.ll
+++ b/llvm/test/CodeGen/AMDGPU/annotate-kernel-features-hsa-call-addrspacecast.ll
@@ -125,10 +125,10 @@ define void @indirect_use_group_to_flat_addrspacecast_queue_ptr() #0 {
attributes #0 = { nounwind }
;.
; GFX8: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
-; GFX8: attributes #[[ATTR1]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; GFX8: attributes #[[ATTR2]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX8: attributes #[[ATTR1]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX8: attributes #[[ATTR2]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
; GFX9: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
-; GFX9: attributes #[[ATTR1]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; GFX9: attributes #[[ATTR2]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX9: attributes #[[ATTR1]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX9: attributes #[[ATTR2]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/annotate-kernel-features-hsa-call.ll b/llvm/test/CodeGen/AMDGPU/annotate-kernel-features-hsa-call.ll
index d625a1fec39bae..42173b3bb4b995 100644
--- a/llvm/test/CodeGen/AMDGPU/annotate-kernel-features-hsa-call.ll
+++ b/llvm/test/CodeGen/AMDGPU/annotate-kernel-features-hsa-call.ll
@@ -392,7 +392,7 @@ define void @func_call_defined() #1 {
}
define void @func_call_asm() #1 {
; ATTRIBUTOR_HSA-LABEL: define {{[^@]+}}@func_call_asm
-; ATTRIBUTOR_HSA-SAME: () #[[ATTR11]] {
+; ATTRIBUTOR_HSA-SAME: () #[[ATTR14:[0-9]+]] {
; ATTRIBUTOR_HSA-NEXT: call void asm sideeffect "", ""() #[[ATTR13]]
; ATTRIBUTOR_HSA-NEXT: ret void
;
@@ -499,7 +499,7 @@ define float @func_other_intrinsic_call(float %arg) #1 {
; Hostcall needs to be enabled for sanitizers
define amdgpu_kernel void @kern_sanitize_address() #2 {
; ATTRIBUTOR_HSA-LABEL: define {{[^@]+}}@kern_sanitize_address
-; ATTRIBUTOR_HSA-SAME: () #[[ATTR15:[0-9]+]] {
+; ATTRIBUTOR_HSA-SAME: () #[[ATTR16:[0-9]+]] {
; ATTRIBUTOR_HSA-NEXT: store volatile i32 0, ptr addrspace(1) null, align 4
; ATTRIBUTOR_HSA-NEXT: ret void
;
@@ -510,7 +510,7 @@ define amdgpu_kernel void @kern_sanitize_address() #2 {
; Hostcall needs to be enabled for sanitizers
define void @func_sanitize_address() #2 {
; ATTRIBUTOR_HSA-LABEL: define {{[^@]+}}@func_sanitize_address
-; ATTRIBUTOR_HSA-SAME: () #[[ATTR15]] {
+; ATTRIBUTOR_HSA-SAME: () #[[ATTR16]] {
; ATTRIBUTOR_HSA-NEXT: store volatile i32 0, ptr addrspace(1) null, align 4
; ATTRIBUTOR_HSA-NEXT: ret void
;
@@ -521,7 +521,7 @@ define void @func_sanitize_address() #2 {
; Hostcall needs to be enabled for sanitizers
define void @func_indirect_sanitize_address() #1 {
; ATTRIBUTOR_HSA-LABEL: define {{[^@]+}}@func_indirect_sanitize_address
-; ATTRIBUTOR_HSA-SAME: () #[[ATTR16:[0-9]+]] {
+; ATTRIBUTOR_HSA-SAME: () #[[ATTR17:[0-9]+]] {
; ATTRIBUTOR_HSA-NEXT: call void @func_sanitize_address()
; ATTRIBUTOR_HSA-NEXT: ret void
;
@@ -532,7 +532,7 @@ define void @func_indirect_sanitize_address() #1 {
; Hostcall needs to be enabled for sanitizers
define amdgpu_kernel void @kern_indirect_sanitize_address() #1 {
; ATTRIBUTOR_HSA-LABEL: define {{[^@]+}}@kern_indirect_sanitize_address
-; ATTRIBUTOR_HSA-SAME: () #[[ATTR16]] {
+; ATTRIBUTOR_HSA-SAME: () #[[ATTR17]] {
; ATTRIBUTOR_HSA-NEXT: call void @func_sanitize_address()
; ATTRIBUTOR_HSA-NEXT: ret void
;
@@ -558,7 +558,7 @@ declare void @enqueue_block_decl() #4
define internal void @enqueue_block_def() #4 {
; ATTRIBUTOR_HSA-LABEL: define {{[^@]+}}@enqueue_block_def
-; ATTRIBUTOR_HSA-SAME: () #[[ATTR19:[0-9]+]] {
+; ATTRIBUTOR_HSA-SAME: () #[[ATTR20:[0-9]+]] {
; ATTRIBUTOR_HSA-NEXT: ret void
;
ret void
@@ -575,7 +575,7 @@ define amdgpu_kernel void @kern_call_enqueued_block_decl() {
define amdgpu_kernel void @kern_call_enqueued_block_def() {
; ATTRIBUTOR_HSA-LABEL: define {{[^@]+}}@kern_call_enqueued_block_def
-; ATTRIBUTOR_HSA-SAME: () #[[ATTR20:[0-9]+]] {
+; ATTRIBUTOR_HSA-SAME: () #[[ATTR21:[0-9]+]] {
; ATTRIBUTOR_HSA-NEXT: call void @enqueue_block_def()
; ATTRIBUTOR_HSA-NEXT: ret void
;
@@ -585,7 +585,7 @@ define amdgpu_kernel void @kern_call_enqueued_block_def() {
define void @unused_enqueue_block() {
; ATTRIBUTOR_HSA-LABEL: define {{[^@]+}}@unused_enqueue_block
-; ATTRIBUTOR_HSA-SAME: () #[[ATTR20]] {
+; ATTRIBUTOR_HSA-SAME: () #[[ATTR21]] {
; ATTRIBUTOR_HSA-NEXT: ret void
;
ret void
@@ -593,7 +593,7 @@ define void @unused_enqueue_block() {
define internal void @known_func() {
; ATTRIBUTOR_HSA-LABEL: define {{[^@]+}}@known_func
-; ATTRIBUTOR_HSA-SAME: () #[[ATTR20]] {
+; ATTRIBUTOR_HSA-SAME: () #[[ATTR21]] {
; ATTRIBUTOR_HSA-NEXT: ret void
;
ret void
@@ -602,8 +602,8 @@ define internal void @known_func() {
; Should never happen
define amdgpu_kernel void @kern_callsite_enqueue_block() {
; ATTRIBUTOR_HSA-LABEL: define {{[^@]+}}@kern_callsite_enqueue_block
-; ATTRIBUTOR_HSA-SAME: () #[[ATTR20]] {
-; ATTRIBUTOR_HSA-NEXT: call void @known_func() #[[ATTR18:[0-9]+]]
+; ATTRIBUTOR_HSA-SAME: () #[[ATTR21]] {
+; ATTRIBUTOR_HSA-NEXT: call void @known_func() #[[ATTR19:[0-9]+]]
; ATTRIBUTOR_HSA-NEXT: ret void
;
call void @known_func() #4
@@ -619,24 +619,25 @@ attributes #4 = { "enqueued-block" }
;.
; ATTRIBUTOR_HSA: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
-; ATTRIBUTOR_HSA: attributes #[[ATTR1]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR2]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR3]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR4]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR5]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR6]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR7]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR8]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR9]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR10]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR11]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR12]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR1]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR2]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR3]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR4]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR5]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR6]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR7]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR8]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR9]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR10]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR11]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR12]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
; ATTRIBUTOR_HSA: attributes #[[ATTR13]] = { nounwind }
-; ATTRIBUTOR_HSA: attributes #[[ATTR14:[0-9]+]] = { nocallback nocreateundeforpoison nofree nosync nounwind speculatable willreturn memory(none) }
-; ATTRIBUTOR_HSA: attributes #[[ATTR15]] = { nounwind sanitize_address "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR16]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR17:[0-9]+]] = { nounwind sanitize_address "amdgpu-no-implicitarg-ptr" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR18]] = { "enqueued-block" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR19]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "enqueued-block" }
-; ATTRIBUTOR_HSA: attributes #[[ATTR20]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR14]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR15:[0-9]+]] = { nocallback nocreateundeforpoison nofree nosync nounwind speculatable willreturn memory(none) }
+; ATTRIBUTOR_HSA: attributes #[[ATTR16]] = { nounwind sanitize_address "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR17]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR18:[0-9]+]] = { nounwind sanitize_address "amdgpu-no-implicitarg-ptr" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR19]] = { "enqueued-block" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR20]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "enqueued-block" }
+; ATTRIBUTOR_HSA: attributes #[[ATTR21]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/annotate-kernel-features-hsa.ll b/llvm/test/CodeGen/AMDGPU/annotate-kernel-features-hsa.ll
index 90d32d0cc48e9c..d077ac448c3796 100644
--- a/llvm/test/CodeGen/AMDGPU/annotate-kernel-features-hsa.ll
+++ b/llvm/test/CodeGen/AMDGPU/annotate-kernel-features-hsa.ll
@@ -475,19 +475,19 @@ attributes #1 = { nounwind }
;.
; HSA: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
; HSA: attributes #[[ATTR1:[0-9]+]] = { nocallback nocreateundeforpoison nofree nosync nounwind speculatable willreturn memory(none) }
-; HSA: attributes #[[ATTR2]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR3]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR4]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR5]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR6]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR7]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR8]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR9]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR10]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workitem-id-x" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR11]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR12]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR13]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; HSA: attributes #[[ATTR14]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR2]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR3]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR4]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR5]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR6]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR7]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR8]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR9]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR10]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workitem-id-x" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR11]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR12]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR13]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; HSA: attributes #[[ATTR14]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
; HSA: [[META0]] = !{i32 1, i32 3, i32 4, i32 16}
; HSA: [[META1]] = !{i32 1, i32 5, i32 6, i32 16}
diff --git a/llvm/test/CodeGen/AMDGPU/annotate-kernel-features.ll b/llvm/test/CodeGen/AMDGPU/annotate-kernel-features.ll
index 2951ad85cc87ec..1757d274d545fa 100644
--- a/llvm/test/CodeGen/AMDGPU/annotate-kernel-features.ll
+++ b/llvm/test/CodeGen/AMDGPU/annotate-kernel-features.ll
@@ -294,13 +294,13 @@ attributes #1 = { nounwind }
;.
; CHECK: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
-; CHECK: attributes #[[ATTR1]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR2]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR3]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR4]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR5]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR6]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR7]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR8]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR9]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workitem-id-x" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR1]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR2]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR3]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR4]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR5]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR6]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR7]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR8]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR9]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workitem-id-x" "amdgpu-no-wwm" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/attributor-flatscratchinit-undefined-behavior.ll b/llvm/test/CodeGen/AMDGPU/attributor-flatscratchinit-undefined-behavior.ll
index fbf85dc4497f97..016ac3594b2717 100644
--- a/llvm/test/CodeGen/AMDGPU/attributor-flatscratchinit-undefined-behavior.ll
+++ b/llvm/test/CodeGen/AMDGPU/attributor-flatscratchinit-undefined-behavior.ll
@@ -147,9 +147,9 @@ define amdgpu_kernel void @call_calls_intrin_ascast_cc_kernel(ptr addrspace(3) %
attributes #0 = { "amdgpu-no-flat-scratch-init" }
;.
-; GFX9: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX9: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
-; GFX10: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX10: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
; GFX9: [[META0]] = !{i32 1, i32 5, i32 6, i32 16}
; GFX9: [[META1]] = !{i32 1, i32 3, i32 4, i32 16}
diff --git a/llvm/test/CodeGen/AMDGPU/attributor-flatscratchinit.ll b/llvm/test/CodeGen/AMDGPU/attributor-flatscratchinit.ll
index 613a937f5b6290..58326d319b8d0c 100644
--- a/llvm/test/CodeGen/AMDGPU/attributor-flatscratchinit.ll
+++ b/llvm/test/CodeGen/AMDGPU/attributor-flatscratchinit.ll
@@ -843,12 +843,12 @@ define amdgpu_kernel void @calls_intrin_ascast_cc_kernel(ptr addrspace(3) %ptr)
define amdgpu_kernel void @with_inline_asm() {
; GFX9-LABEL: define amdgpu_kernel void @with_inline_asm(
-; GFX9-SAME: ) #[[ATTR0]] {
+; GFX9-SAME: ) #[[ATTR4:[0-9]+]] {
; GFX9-NEXT: call void asm sideeffect "
; GFX9-NEXT: ret void
;
; GFX10-LABEL: define amdgpu_kernel void @with_inline_asm(
-; GFX10-SAME: ) #[[ATTR0]] {
+; GFX10-SAME: ) #[[ATTR4:[0-9]+]] {
; GFX10-NEXT: call void asm sideeffect "
; GFX10-NEXT: ret void
;
@@ -857,15 +857,17 @@ define amdgpu_kernel void @with_inline_asm() {
}
;.
-; GFX9: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; GFX9: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX9: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX9: attributes #[[ATTR1]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
; GFX9: attributes #[[ATTR2:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
-; GFX9: attributes #[[ATTR3]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX9: attributes #[[ATTR3]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX9: attributes #[[ATTR4]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
-; GFX10: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; GFX10: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX10: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX10: attributes #[[ATTR1]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
; GFX10: attributes #[[ATTR2:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
-; GFX10: attributes #[[ATTR3]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX10: attributes #[[ATTR3]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; GFX10: attributes #[[ATTR4]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
; GFX9: [[META0]] = !{i32 2, i32 16}
; GFX9: [[META1]] = !{i32 1, i32 2, i32 3, i32 16}
diff --git a/llvm/test/CodeGen/AMDGPU/attributor-wwm.ll b/llvm/test/CodeGen/AMDGPU/attributor-wwm.ll
index 23bdc48e1e8d59..ab2832621ec9b6 100644
--- a/llvm/test/CodeGen/AMDGPU/attributor-wwm.ll
+++ b/llvm/test/CodeGen/AMDGPU/attributor-wwm.ll
@@ -58,7 +58,7 @@ define amdgpu_kernel void @test_nested(i32 %input, ptr addrspace(1) %out) {
ret void
}
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" }
; CHECK: attributes #[[ATTR2:[0-9]+]] = { convergent nocallback nofree nounwind speculatable willreturn memory(none) }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/direct-indirect-call.ll b/llvm/test/CodeGen/AMDGPU/direct-indirect-call.ll
index 085628c50e3e0d..62230d08ba39f1 100644
--- a/llvm/test/CodeGen/AMDGPU/direct-indirect-call.ll
+++ b/llvm/test/CodeGen/AMDGPU/direct-indirect-call.ll
@@ -33,5 +33,5 @@ define amdgpu_kernel void @test_direct_indirect_call() {
ret void
}
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/duplicate-attribute-indirect.ll b/llvm/test/CodeGen/AMDGPU/duplicate-attribute-indirect.ll
index 18ab7542fa3a06..04713d383a6d27 100644
--- a/llvm/test/CodeGen/AMDGPU/duplicate-attribute-indirect.ll
+++ b/llvm/test/CodeGen/AMDGPU/duplicate-attribute-indirect.ll
@@ -28,6 +28,6 @@ define amdgpu_kernel void @test_simple_indirect_call() #0 {
attributes #0 = { "amdgpu-no-dispatch-id" }
;.
-; ATTRIBUTOR_GCN: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_GCN: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
; ATTRIBUTOR_GCN: attributes #[[ATTR1]] = { "amdgpu-no-dispatch-id" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll b/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll
index d83c8d133e7276..f40b209b77da7a 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll
@@ -6750,6 +6750,7 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_max_i32 v[0:1], v2
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_max_saddr_i32_nortn:
@@ -6763,6 +6764,7 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_i32 v[2:3], v1
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_max_saddr_i32_nortn:
@@ -6771,6 +6773,7 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: v_mov_b32_e32 v1, 0
; GFX950-SDAG-NEXT: v_lshl_add_u64 v[0:1], s[2:3], 0, v[0:1]
; GFX950-SDAG-NEXT: flat_atomic_smax v[0:1], v2
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_max_saddr_i32_nortn:
@@ -6780,6 +6783,7 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, 0, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_smax v[2:3], v1
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -6798,6 +6802,7 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_max_i32 v[0:1], v2 offset:-128
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_max_saddr_i32_nortn_neg128:
@@ -6811,6 +6816,7 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_i32 v[2:3], v1 offset:-128
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_max_saddr_i32_nortn_neg128:
@@ -6822,6 +6828,7 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_addc_co_u32_e32 v1, vcc, -1, v1, vcc
; GFX950-SDAG-NEXT: flat_atomic_smax v[0:1], v2
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_max_saddr_i32_nortn_neg128:
@@ -6834,6 +6841,7 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, -1, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_smax v[2:3], v1
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -6865,18 +6873,16 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB58_4
; GFX1250-SDAG-NEXT: .LBB58_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB58_5
; GFX1250-SDAG-NEXT: .LBB58_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_max_i64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB58_2
; GFX1250-SDAG-NEXT: .LBB58_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -6914,18 +6920,16 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB58_4
; GFX1250-GISEL-NEXT: .LBB58_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB58_5
; GFX1250-GISEL-NEXT: .LBB58_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_max_i64 v[0:1], v[2:3], v[4:5] th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB58_2
; GFX1250-GISEL-NEXT: .LBB58_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -6956,10 +6960,11 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB58_4
; GFX950-SDAG-NEXT: .LBB58_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB58_5
; GFX950-SDAG-NEXT: .LBB58_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smax_x2 v[0:1], v[4:5], v[2:3] sc0
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -6968,7 +6973,6 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_gt_i64_e32 vcc, v[0:1], v[2:3]
@@ -7000,10 +7004,11 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB58_4
; GFX950-GISEL-NEXT: .LBB58_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB58_5
; GFX950-GISEL-NEXT: .LBB58_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smax_x2 v[0:1], v[2:3], v[4:5] sc0
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7012,7 +7017,6 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_gt_i64_e32 vcc, v[0:1], v[4:5]
@@ -7057,18 +7061,16 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB59_4
; GFX1250-SDAG-NEXT: .LBB59_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB59_5
; GFX1250-SDAG-NEXT: .LBB59_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_max_i64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB59_2
; GFX1250-SDAG-NEXT: .LBB59_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -7109,18 +7111,16 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB59_4
; GFX1250-GISEL-NEXT: .LBB59_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB59_5
; GFX1250-GISEL-NEXT: .LBB59_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_max_i64 v[0:1], v[6:7], v[4:5] offset:-128 th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB59_2
; GFX1250-GISEL-NEXT: .LBB59_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -7154,10 +7154,11 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB59_4
; GFX950-SDAG-NEXT: .LBB59_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB59_5
; GFX950-SDAG-NEXT: .LBB59_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smax_x2 v[0:1], v[4:5], v[2:3] sc0
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7166,7 +7167,6 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_gt_i64_e32 vcc, v[0:1], v[2:3]
@@ -7201,10 +7201,11 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB59_4
; GFX950-GISEL-NEXT: .LBB59_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB59_5
; GFX950-GISEL-NEXT: .LBB59_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smax_x2 v[0:1], v[2:3], v[4:5] sc0
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7213,7 +7214,6 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_gt_i64_e32 vcc, v[0:1], v[4:5]
@@ -7259,6 +7259,7 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: flat_atomic_max_i64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB60_2
@@ -7300,6 +7301,7 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: flat_atomic_max_i64 v[0:1], v[4:5]
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB60_2
@@ -7333,6 +7335,7 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB60_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smax_x2 v[0:1], v[2:3]
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7369,6 +7372,7 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB60_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smax_x2 v[0:1], v[4:5]
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7419,6 +7423,7 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: flat_atomic_max_i64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB61_2
@@ -7463,6 +7468,7 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: flat_atomic_max_i64 v[2:3], v[4:5] offset:-128
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB61_2
@@ -7499,6 +7505,7 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB61_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smax_x2 v[0:1], v[2:3]
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7539,6 +7546,7 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB61_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smax_x2 v[0:1], v[4:5]
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7690,6 +7698,7 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_min_i32 v[0:1], v2
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_min_saddr_i32_nortn:
@@ -7703,6 +7712,7 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_i32 v[2:3], v1
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_min_saddr_i32_nortn:
@@ -7711,6 +7721,7 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: v_mov_b32_e32 v1, 0
; GFX950-SDAG-NEXT: v_lshl_add_u64 v[0:1], s[2:3], 0, v[0:1]
; GFX950-SDAG-NEXT: flat_atomic_smin v[0:1], v2
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_min_saddr_i32_nortn:
@@ -7720,6 +7731,7 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, 0, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_smin v[2:3], v1
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -7738,6 +7750,7 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_min_i32 v[0:1], v2 offset:-128
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_min_saddr_i32_nortn_neg128:
@@ -7751,6 +7764,7 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_i32 v[2:3], v1 offset:-128
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_min_saddr_i32_nortn_neg128:
@@ -7762,6 +7776,7 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_addc_co_u32_e32 v1, vcc, -1, v1, vcc
; GFX950-SDAG-NEXT: flat_atomic_smin v[0:1], v2
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_min_saddr_i32_nortn_neg128:
@@ -7774,6 +7789,7 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, -1, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_smin v[2:3], v1
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -7805,18 +7821,16 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB66_4
; GFX1250-SDAG-NEXT: .LBB66_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB66_5
; GFX1250-SDAG-NEXT: .LBB66_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_min_i64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB66_2
; GFX1250-SDAG-NEXT: .LBB66_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -7854,18 +7868,16 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB66_4
; GFX1250-GISEL-NEXT: .LBB66_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB66_5
; GFX1250-GISEL-NEXT: .LBB66_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_min_i64 v[0:1], v[2:3], v[4:5] th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB66_2
; GFX1250-GISEL-NEXT: .LBB66_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -7896,10 +7908,11 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB66_4
; GFX950-SDAG-NEXT: .LBB66_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB66_5
; GFX950-SDAG-NEXT: .LBB66_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smin_x2 v[0:1], v[4:5], v[2:3] sc0
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7908,7 +7921,6 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_le_i64_e32 vcc, v[0:1], v[2:3]
@@ -7940,10 +7952,11 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB66_4
; GFX950-GISEL-NEXT: .LBB66_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB66_5
; GFX950-GISEL-NEXT: .LBB66_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smin_x2 v[0:1], v[2:3], v[4:5] sc0
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -7952,7 +7965,6 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_lt_i64_e32 vcc, v[0:1], v[4:5]
@@ -7997,18 +8009,16 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB67_4
; GFX1250-SDAG-NEXT: .LBB67_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB67_5
; GFX1250-SDAG-NEXT: .LBB67_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_min_i64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB67_2
; GFX1250-SDAG-NEXT: .LBB67_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -8049,18 +8059,16 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB67_4
; GFX1250-GISEL-NEXT: .LBB67_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB67_5
; GFX1250-GISEL-NEXT: .LBB67_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_min_i64 v[0:1], v[6:7], v[4:5] offset:-128 th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB67_2
; GFX1250-GISEL-NEXT: .LBB67_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -8094,10 +8102,11 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB67_4
; GFX950-SDAG-NEXT: .LBB67_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB67_5
; GFX950-SDAG-NEXT: .LBB67_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smin_x2 v[0:1], v[4:5], v[2:3] sc0
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8106,7 +8115,6 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_le_i64_e32 vcc, v[0:1], v[2:3]
@@ -8141,10 +8149,11 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB67_4
; GFX950-GISEL-NEXT: .LBB67_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB67_5
; GFX950-GISEL-NEXT: .LBB67_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smin_x2 v[0:1], v[2:3], v[4:5] sc0
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8153,7 +8162,6 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_lt_i64_e32 vcc, v[0:1], v[4:5]
@@ -8199,6 +8207,7 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: flat_atomic_min_i64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB68_2
@@ -8240,6 +8249,7 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: flat_atomic_min_i64 v[0:1], v[4:5]
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB68_2
@@ -8273,6 +8283,7 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB68_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smin_x2 v[0:1], v[2:3]
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8309,6 +8320,7 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB68_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smin_x2 v[0:1], v[4:5]
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8359,6 +8371,7 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: flat_atomic_min_i64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB69_2
@@ -8403,6 +8416,7 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: flat_atomic_min_i64 v[2:3], v[4:5] offset:-128
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB69_2
@@ -8439,6 +8453,7 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB69_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_smin_x2 v[0:1], v[2:3]
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8479,6 +8494,7 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB69_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_smin_x2 v[0:1], v[4:5]
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8630,6 +8646,7 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_max_u32 v[0:1], v2
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_umax_saddr_i32_nortn:
@@ -8643,6 +8660,7 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_u32 v[2:3], v1
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_umax_saddr_i32_nortn:
@@ -8651,6 +8669,7 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: v_mov_b32_e32 v1, 0
; GFX950-SDAG-NEXT: v_lshl_add_u64 v[0:1], s[2:3], 0, v[0:1]
; GFX950-SDAG-NEXT: flat_atomic_umax v[0:1], v2
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_umax_saddr_i32_nortn:
@@ -8660,6 +8679,7 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, 0, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_umax v[2:3], v1
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -8678,6 +8698,7 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_max_u32 v[0:1], v2 offset:-128
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_umax_saddr_i32_nortn_neg128:
@@ -8691,6 +8712,7 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_u32 v[2:3], v1 offset:-128
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_umax_saddr_i32_nortn_neg128:
@@ -8702,6 +8724,7 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_addc_co_u32_e32 v1, vcc, -1, v1, vcc
; GFX950-SDAG-NEXT: flat_atomic_umax v[0:1], v2
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_umax_saddr_i32_nortn_neg128:
@@ -8714,6 +8737,7 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, -1, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_umax v[2:3], v1
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -8745,18 +8769,16 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB74_4
; GFX1250-SDAG-NEXT: .LBB74_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB74_5
; GFX1250-SDAG-NEXT: .LBB74_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_max_u64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB74_2
; GFX1250-SDAG-NEXT: .LBB74_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -8794,18 +8816,16 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB74_4
; GFX1250-GISEL-NEXT: .LBB74_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB74_5
; GFX1250-GISEL-NEXT: .LBB74_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_max_u64 v[0:1], v[2:3], v[4:5] th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB74_2
; GFX1250-GISEL-NEXT: .LBB74_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -8836,10 +8856,11 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB74_4
; GFX950-SDAG-NEXT: .LBB74_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB74_5
; GFX950-SDAG-NEXT: .LBB74_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umax_x2 v[0:1], v[4:5], v[2:3] sc0
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8848,7 +8869,6 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_gt_u64_e32 vcc, v[0:1], v[2:3]
@@ -8880,10 +8900,11 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB74_4
; GFX950-GISEL-NEXT: .LBB74_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB74_5
; GFX950-GISEL-NEXT: .LBB74_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umax_x2 v[0:1], v[2:3], v[4:5] sc0
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -8892,7 +8913,6 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_gt_u64_e32 vcc, v[0:1], v[4:5]
@@ -8937,18 +8957,16 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB75_4
; GFX1250-SDAG-NEXT: .LBB75_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB75_5
; GFX1250-SDAG-NEXT: .LBB75_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_max_u64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB75_2
; GFX1250-SDAG-NEXT: .LBB75_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -8989,18 +9007,16 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB75_4
; GFX1250-GISEL-NEXT: .LBB75_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB75_5
; GFX1250-GISEL-NEXT: .LBB75_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_max_u64 v[0:1], v[6:7], v[4:5] offset:-128 th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB75_2
; GFX1250-GISEL-NEXT: .LBB75_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -9034,10 +9050,11 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB75_4
; GFX950-SDAG-NEXT: .LBB75_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB75_5
; GFX950-SDAG-NEXT: .LBB75_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umax_x2 v[0:1], v[4:5], v[2:3] sc0
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9046,7 +9063,6 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_gt_u64_e32 vcc, v[0:1], v[2:3]
@@ -9081,10 +9097,11 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB75_4
; GFX950-GISEL-NEXT: .LBB75_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB75_5
; GFX950-GISEL-NEXT: .LBB75_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umax_x2 v[0:1], v[2:3], v[4:5] sc0
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9093,7 +9110,6 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_gt_u64_e32 vcc, v[0:1], v[4:5]
@@ -9139,6 +9155,7 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: flat_atomic_max_u64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB76_2
@@ -9180,6 +9197,7 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: flat_atomic_max_u64 v[0:1], v[4:5]
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB76_2
@@ -9213,6 +9231,7 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB76_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umax_x2 v[0:1], v[2:3]
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9249,6 +9268,7 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB76_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umax_x2 v[0:1], v[4:5]
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9299,6 +9319,7 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-SDAG-NEXT: flat_atomic_max_u64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB77_2
@@ -9343,6 +9364,7 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: flat_atomic_max_u64 v[2:3], v[4:5] offset:-128
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB77_2
@@ -9379,6 +9401,7 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB77_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umax_x2 v[0:1], v[2:3]
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9419,6 +9442,7 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB77_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umax_x2 v[0:1], v[4:5]
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9570,6 +9594,7 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_min_u32 v[0:1], v2
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_umin_saddr_i32_nortn:
@@ -9583,6 +9608,7 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_u32 v[2:3], v1
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_umin_saddr_i32_nortn:
@@ -9591,6 +9617,7 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: v_mov_b32_e32 v1, 0
; GFX950-SDAG-NEXT: v_lshl_add_u64 v[0:1], s[2:3], 0, v[0:1]
; GFX950-SDAG-NEXT: flat_atomic_umin v[0:1], v2
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_umin_saddr_i32_nortn:
@@ -9600,6 +9627,7 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, 0, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_umin v[2:3], v1
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -9618,6 +9646,7 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: flat_atomic_min_u32 v[0:1], v2 offset:-128
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_endpgm
;
; GFX1250-GISEL-LABEL: flat_umin_saddr_i32_nortn_neg128:
@@ -9631,6 +9660,7 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_u32 v[2:3], v1 offset:-128
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_endpgm
;
; GFX950-SDAG-LABEL: flat_umin_saddr_i32_nortn_neg128:
@@ -9642,6 +9672,7 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_addc_co_u32_e32 v1, vcc, -1, v1, vcc
; GFX950-SDAG-NEXT: flat_atomic_umin v[0:1], v2
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: s_endpgm
;
; GFX950-GISEL-LABEL: flat_umin_saddr_i32_nortn_neg128:
@@ -9654,6 +9685,7 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_addc_co_u32_e32 v3, vcc, -1, v3, vcc
; GFX950-GISEL-NEXT: flat_atomic_umin v[2:3], v1
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr %sbase, i64 %zext.offset
@@ -9685,18 +9717,16 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB82_4
; GFX1250-SDAG-NEXT: .LBB82_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB82_5
; GFX1250-SDAG-NEXT: .LBB82_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_min_u64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB82_2
; GFX1250-SDAG-NEXT: .LBB82_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -9734,18 +9764,16 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB82_4
; GFX1250-GISEL-NEXT: .LBB82_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB82_5
; GFX1250-GISEL-NEXT: .LBB82_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_min_u64 v[0:1], v[2:3], v[4:5] th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB82_2
; GFX1250-GISEL-NEXT: .LBB82_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -9776,10 +9804,11 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB82_4
; GFX950-SDAG-NEXT: .LBB82_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB82_5
; GFX950-SDAG-NEXT: .LBB82_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umin_x2 v[0:1], v[4:5], v[2:3] sc0
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9788,7 +9817,6 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_le_u64_e32 vcc, v[0:1], v[2:3]
@@ -9820,10 +9848,11 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB82_4
; GFX950-GISEL-NEXT: .LBB82_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB82_5
; GFX950-GISEL-NEXT: .LBB82_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umin_x2 v[0:1], v[2:3], v[4:5] sc0
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9832,7 +9861,6 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_lt_u64_e32 vcc, v[0:1], v[4:5]
@@ -9877,18 +9905,16 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB83_4
; GFX1250-SDAG-NEXT: .LBB83_2: ; %atomicrmw.phi
; GFX1250-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_branch .LBB83_5
; GFX1250-SDAG-NEXT: .LBB83_3: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: flat_atomic_min_u64 v[0:1], v[4:5], v[2:3] th:TH_ATOMIC_RETURN
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
-; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
+; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB83_2
; GFX1250-SDAG-NEXT: .LBB83_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
@@ -9929,18 +9955,16 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB83_4
; GFX1250-GISEL-NEXT: .LBB83_2: ; %atomicrmw.phi
; GFX1250-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_branch .LBB83_5
; GFX1250-GISEL-NEXT: .LBB83_3: ; %atomicrmw.global
; GFX1250-GISEL-NEXT: flat_atomic_min_u64 v[0:1], v[6:7], v[4:5] offset:-128 th:TH_ATOMIC_RETURN
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
-; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
+; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB83_2
; GFX1250-GISEL-NEXT: .LBB83_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
@@ -9974,10 +9998,11 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-SDAG-NEXT: s_cbranch_execnz .LBB83_4
; GFX950-SDAG-NEXT: .LBB83_2: ; %atomicrmw.phi
; GFX950-SDAG-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: s_branch .LBB83_5
; GFX950-SDAG-NEXT: .LBB83_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umin_x2 v[0:1], v[4:5], v[2:3] sc0
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -9986,7 +10011,6 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-SDAG-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[4:5]
; GFX950-SDAG-NEXT: s_nop 1
; GFX950-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc
-; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: scratch_load_dwordx2 v[0:1], v4, off
; GFX950-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX950-SDAG-NEXT: v_cmp_le_u64_e32 vcc, v[0:1], v[2:3]
@@ -10021,10 +10045,11 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-GISEL-NEXT: s_cbranch_execnz .LBB83_4
; GFX950-GISEL-NEXT: .LBB83_2: ; %atomicrmw.phi
; GFX950-GISEL-NEXT: s_or_b64 exec, exec, s[0:1]
-; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: s_branch .LBB83_5
; GFX950-GISEL-NEXT: .LBB83_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umin_x2 v[0:1], v[2:3], v[4:5] sc0
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -10033,7 +10058,6 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX950-GISEL-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
; GFX950-GISEL-NEXT: s_nop 1
; GFX950-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v2, vcc
-; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: scratch_load_dwordx2 v[0:1], v6, off
; GFX950-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX950-GISEL-NEXT: v_cmp_lt_u64_e32 vcc, v[0:1], v[4:5]
@@ -10079,6 +10103,7 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: flat_atomic_min_u64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB84_2
@@ -10120,6 +10145,7 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: flat_atomic_min_u64 v[0:1], v[4:5]
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB84_2
@@ -10153,6 +10179,7 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB84_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umin_x2 v[0:1], v[2:3]
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -10189,6 +10216,7 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB84_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umin_x2 v[0:1], v[4:5]
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -10239,6 +10267,7 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-SDAG-NEXT: flat_atomic_min_u64 v[0:1], v[2:3]
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1250-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1250-SDAG-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-SDAG-NEXT: s_cbranch_execz .LBB85_2
@@ -10283,6 +10312,7 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: flat_atomic_min_u64 v[2:3], v[4:5] offset:-128
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr0
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX1250-GISEL-NEXT: s_wait_storecnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-GISEL-NEXT: s_cbranch_execz .LBB85_2
@@ -10319,6 +10349,7 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-SDAG-NEXT: s_endpgm
; GFX950-SDAG-NEXT: .LBB85_3: ; %atomicrmw.global
; GFX950-SDAG-NEXT: flat_atomic_umin_x2 v[0:1], v[2:3]
+; GFX950-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-SDAG-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX950-SDAG-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
@@ -10359,6 +10390,7 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX950-GISEL-NEXT: s_endpgm
; GFX950-GISEL-NEXT: .LBB85_3: ; %atomicrmw.global
; GFX950-GISEL-NEXT: flat_atomic_umin_x2 v[0:1], v[4:5]
+; GFX950-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX950-GISEL-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX950-GISEL-NEXT: s_andn2_saveexec_b64 s[0:1], s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/global-saddr-atomics.ll b/llvm/test/CodeGen/AMDGPU/global-saddr-atomics.ll
index e728825cc2f5ef..c9ebcc15242580 100644
--- a/llvm/test/CodeGen/AMDGPU/global-saddr-atomics.ll
+++ b/llvm/test/CodeGen/AMDGPU/global-saddr-atomics.ll
@@ -2142,22 +2142,31 @@ define amdgpu_ps void @global_xor_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; --------------------------------------------------------------------------------
define amdgpu_ps float @global_max_saddr_i32_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_max_saddr_i32_rtn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smax v0, v0, v1, s[2:3] glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_max_saddr_i32_rtn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smax v0, v0, v1, s[2:3] glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_max_saddr_i32_rtn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smax v0, v0, v1, s[2:3] glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_max_saddr_i32_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i32 v0, v0, v1, s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_max_saddr_i32_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_max_i32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2167,22 +2176,31 @@ define amdgpu_ps float @global_max_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps float @global_max_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_max_saddr_i32_rtn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smax v0, v0, v1, s[2:3] offset:-128 glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_max_saddr_i32_rtn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smax v0, v0, v1, s[2:3] offset:-128 glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_max_saddr_i32_rtn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smax v0, v0, v1, s[2:3] offset:-128 glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_max_saddr_i32_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i32 v0, v0, v1, s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_max_saddr_i32_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_max_i32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2193,19 +2211,30 @@ define amdgpu_ps float @global_max_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %
}
define amdgpu_ps void @global_max_saddr_i32_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_max_saddr_i32_nortn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smax v0, v1, s[2:3]
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_max_saddr_i32_nortn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smax v0, v1, s[2:3]
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_max_saddr_i32_nortn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smax v0, v1, s[2:3]
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_max_saddr_i32_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i32 v0, v1, s[2:3]
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_max_saddr_i32_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i32 v0, v1, s[2:3]
+; GFX12-NEXT: global_atomic_max_i32 v0, v1, s[2:3] scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2214,19 +2243,30 @@ define amdgpu_ps void @global_max_saddr_i32_nortn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps void @global_max_saddr_i32_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_max_saddr_i32_nortn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smax v0, v1, s[2:3] offset:-128
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_max_saddr_i32_nortn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smax v0, v1, s[2:3] offset:-128
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_max_saddr_i32_nortn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smax v0, v1, s[2:3] offset:-128
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_max_saddr_i32_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i32 v0, v1, s[2:3] offset:-128
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_max_saddr_i32_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i32 v0, v1, s[2:3] offset:-128
+; GFX12-NEXT: global_atomic_max_i32 v0, v1, s[2:3] offset:-128 scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2236,22 +2276,31 @@ define amdgpu_ps void @global_max_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
}
define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_max_saddr_i64_rtn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smax_x2 v[0:1], v0, v[1:2], s[2:3] glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_max_saddr_i64_rtn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smax_x2 v[0:1], v0, v[1:2], s[2:3] glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_max_saddr_i64_rtn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smax_x2 v[0:1], v0, v[1:2], s[2:3] glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_max_saddr_i64_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i64 v[0:1], v0, v[1:2], s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_max_saddr_i64_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_max_i64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2261,22 +2310,31 @@ define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn(ptr addrspace(1) inreg %s
}
define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_max_saddr_i64_rtn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smax_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_max_saddr_i64_rtn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smax_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_max_saddr_i64_rtn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smax_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_max_saddr_i64_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i64 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_max_saddr_i64_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_max_i64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2287,19 +2345,30 @@ define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn_neg128(ptr addrspace(1) i
}
define amdgpu_ps void @global_max_saddr_i64_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_max_saddr_i64_nortn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smax_x2 v0, v[1:2], s[2:3]
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_max_saddr_i64_nortn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smax_x2 v0, v[1:2], s[2:3]
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_max_saddr_i64_nortn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smax_x2 v0, v[1:2], s[2:3]
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_max_saddr_i64_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i64 v0, v[1:2], s[2:3]
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_max_saddr_i64_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i64 v0, v[1:2], s[2:3]
+; GFX12-NEXT: global_atomic_max_i64 v0, v[1:2], s[2:3] scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2308,19 +2377,30 @@ define amdgpu_ps void @global_max_saddr_i64_nortn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps void @global_max_saddr_i64_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_max_saddr_i64_nortn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smax_x2 v0, v[1:2], s[2:3] offset:-128
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_max_saddr_i64_nortn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smax_x2 v0, v[1:2], s[2:3] offset:-128
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_max_saddr_i64_nortn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smax_x2 v0, v[1:2], s[2:3] offset:-128
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_max_saddr_i64_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_i64 v0, v[1:2], s[2:3] offset:-128
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_max_saddr_i64_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_i64 v0, v[1:2], s[2:3] offset:-128
+; GFX12-NEXT: global_atomic_max_i64 v0, v[1:2], s[2:3] offset:-128 scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2334,22 +2414,31 @@ define amdgpu_ps void @global_max_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; --------------------------------------------------------------------------------
define amdgpu_ps float @global_min_saddr_i32_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_min_saddr_i32_rtn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smin v0, v0, v1, s[2:3] glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_min_saddr_i32_rtn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smin v0, v0, v1, s[2:3] glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_min_saddr_i32_rtn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smin v0, v0, v1, s[2:3] glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_min_saddr_i32_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i32 v0, v0, v1, s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_min_saddr_i32_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_min_i32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2359,22 +2448,31 @@ define amdgpu_ps float @global_min_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps float @global_min_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_min_saddr_i32_rtn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smin v0, v0, v1, s[2:3] offset:-128 glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_min_saddr_i32_rtn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smin v0, v0, v1, s[2:3] offset:-128 glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_min_saddr_i32_rtn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smin v0, v0, v1, s[2:3] offset:-128 glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_min_saddr_i32_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i32 v0, v0, v1, s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_min_saddr_i32_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_min_i32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2385,19 +2483,30 @@ define amdgpu_ps float @global_min_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %
}
define amdgpu_ps void @global_min_saddr_i32_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_min_saddr_i32_nortn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smin v0, v1, s[2:3]
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_min_saddr_i32_nortn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smin v0, v1, s[2:3]
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_min_saddr_i32_nortn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smin v0, v1, s[2:3]
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_min_saddr_i32_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i32 v0, v1, s[2:3]
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_min_saddr_i32_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i32 v0, v1, s[2:3]
+; GFX12-NEXT: global_atomic_min_i32 v0, v1, s[2:3] scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2406,19 +2515,30 @@ define amdgpu_ps void @global_min_saddr_i32_nortn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps void @global_min_saddr_i32_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_min_saddr_i32_nortn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smin v0, v1, s[2:3] offset:-128
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_min_saddr_i32_nortn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smin v0, v1, s[2:3] offset:-128
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_min_saddr_i32_nortn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smin v0, v1, s[2:3] offset:-128
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_min_saddr_i32_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i32 v0, v1, s[2:3] offset:-128
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_min_saddr_i32_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i32 v0, v1, s[2:3] offset:-128
+; GFX12-NEXT: global_atomic_min_i32 v0, v1, s[2:3] offset:-128 scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2428,22 +2548,31 @@ define amdgpu_ps void @global_min_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
}
define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_min_saddr_i64_rtn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smin_x2 v[0:1], v0, v[1:2], s[2:3] glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_min_saddr_i64_rtn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smin_x2 v[0:1], v0, v[1:2], s[2:3] glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_min_saddr_i64_rtn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smin_x2 v[0:1], v0, v[1:2], s[2:3] glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_min_saddr_i64_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i64 v[0:1], v0, v[1:2], s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_min_saddr_i64_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_min_i64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2453,22 +2582,31 @@ define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn(ptr addrspace(1) inreg %s
}
define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_min_saddr_i64_rtn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smin_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_min_saddr_i64_rtn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smin_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_min_saddr_i64_rtn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smin_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_min_saddr_i64_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i64 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_min_saddr_i64_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_min_i64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2479,19 +2617,30 @@ define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn_neg128(ptr addrspace(1) i
}
define amdgpu_ps void @global_min_saddr_i64_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_min_saddr_i64_nortn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smin_x2 v0, v[1:2], s[2:3]
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_min_saddr_i64_nortn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smin_x2 v0, v[1:2], s[2:3]
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_min_saddr_i64_nortn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smin_x2 v0, v[1:2], s[2:3]
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_min_saddr_i64_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i64 v0, v[1:2], s[2:3]
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_min_saddr_i64_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i64 v0, v[1:2], s[2:3]
+; GFX12-NEXT: global_atomic_min_i64 v0, v[1:2], s[2:3] scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2500,19 +2649,30 @@ define amdgpu_ps void @global_min_saddr_i64_nortn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps void @global_min_saddr_i64_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_min_saddr_i64_nortn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_smin_x2 v0, v[1:2], s[2:3] offset:-128
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_min_saddr_i64_nortn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_smin_x2 v0, v[1:2], s[2:3] offset:-128
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_min_saddr_i64_nortn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_smin_x2 v0, v[1:2], s[2:3] offset:-128
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_min_saddr_i64_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_i64 v0, v[1:2], s[2:3] offset:-128
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_min_saddr_i64_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_i64 v0, v[1:2], s[2:3] offset:-128
+; GFX12-NEXT: global_atomic_min_i64 v0, v[1:2], s[2:3] offset:-128 scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2526,22 +2686,31 @@ define amdgpu_ps void @global_min_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; --------------------------------------------------------------------------------
define amdgpu_ps float @global_umax_saddr_i32_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_umax_saddr_i32_rtn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umax v0, v0, v1, s[2:3] glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_umax_saddr_i32_rtn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umax v0, v0, v1, s[2:3] glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_umax_saddr_i32_rtn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umax v0, v0, v1, s[2:3] glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umax_saddr_i32_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u32 v0, v0, v1, s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umax_saddr_i32_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_max_u32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2551,22 +2720,31 @@ define amdgpu_ps float @global_umax_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps float @global_umax_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_umax_saddr_i32_rtn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umax v0, v0, v1, s[2:3] offset:-128 glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_umax_saddr_i32_rtn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umax v0, v0, v1, s[2:3] offset:-128 glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_umax_saddr_i32_rtn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umax v0, v0, v1, s[2:3] offset:-128 glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umax_saddr_i32_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u32 v0, v0, v1, s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umax_saddr_i32_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_max_u32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2577,19 +2755,30 @@ define amdgpu_ps float @global_umax_saddr_i32_rtn_neg128(ptr addrspace(1) inreg
}
define amdgpu_ps void @global_umax_saddr_i32_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_umax_saddr_i32_nortn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umax v0, v1, s[2:3]
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_umax_saddr_i32_nortn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umax v0, v1, s[2:3]
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_umax_saddr_i32_nortn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umax v0, v1, s[2:3]
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_umax_saddr_i32_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u32 v0, v1, s[2:3]
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umax_saddr_i32_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u32 v0, v1, s[2:3]
+; GFX12-NEXT: global_atomic_max_u32 v0, v1, s[2:3] scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2598,19 +2787,30 @@ define amdgpu_ps void @global_umax_saddr_i32_nortn(ptr addrspace(1) inreg %sbase
}
define amdgpu_ps void @global_umax_saddr_i32_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_umax_saddr_i32_nortn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umax v0, v1, s[2:3] offset:-128
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_umax_saddr_i32_nortn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umax v0, v1, s[2:3] offset:-128
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_umax_saddr_i32_nortn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umax v0, v1, s[2:3] offset:-128
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_umax_saddr_i32_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u32 v0, v1, s[2:3] offset:-128
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umax_saddr_i32_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u32 v0, v1, s[2:3] offset:-128
+; GFX12-NEXT: global_atomic_max_u32 v0, v1, s[2:3] offset:-128 scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2620,22 +2820,31 @@ define amdgpu_ps void @global_umax_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
}
define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_umax_saddr_i64_rtn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umax_x2 v[0:1], v0, v[1:2], s[2:3] glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_umax_saddr_i64_rtn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umax_x2 v[0:1], v0, v[1:2], s[2:3] glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_umax_saddr_i64_rtn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umax_x2 v[0:1], v0, v[1:2], s[2:3] glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umax_saddr_i64_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u64 v[0:1], v0, v[1:2], s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umax_saddr_i64_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_max_u64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2645,22 +2854,31 @@ define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn(ptr addrspace(1) inreg %
}
define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_umax_saddr_i64_rtn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umax_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_umax_saddr_i64_rtn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umax_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_umax_saddr_i64_rtn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umax_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umax_saddr_i64_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u64 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umax_saddr_i64_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_max_u64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2671,19 +2889,30 @@ define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn_neg128(ptr addrspace(1)
}
define amdgpu_ps void @global_umax_saddr_i64_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_umax_saddr_i64_nortn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umax_x2 v0, v[1:2], s[2:3]
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_umax_saddr_i64_nortn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umax_x2 v0, v[1:2], s[2:3]
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_umax_saddr_i64_nortn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umax_x2 v0, v[1:2], s[2:3]
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_umax_saddr_i64_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u64 v0, v[1:2], s[2:3]
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umax_saddr_i64_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u64 v0, v[1:2], s[2:3]
+; GFX12-NEXT: global_atomic_max_u64 v0, v[1:2], s[2:3] scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2692,19 +2921,30 @@ define amdgpu_ps void @global_umax_saddr_i64_nortn(ptr addrspace(1) inreg %sbase
}
define amdgpu_ps void @global_umax_saddr_i64_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_umax_saddr_i64_nortn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umax_x2 v0, v[1:2], s[2:3] offset:-128
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_umax_saddr_i64_nortn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umax_x2 v0, v[1:2], s[2:3] offset:-128
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_umax_saddr_i64_nortn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umax_x2 v0, v[1:2], s[2:3] offset:-128
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_umax_saddr_i64_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_max_u64 v0, v[1:2], s[2:3] offset:-128
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umax_saddr_i64_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_max_u64 v0, v[1:2], s[2:3] offset:-128
+; GFX12-NEXT: global_atomic_max_u64 v0, v[1:2], s[2:3] offset:-128 scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2718,22 +2958,31 @@ define amdgpu_ps void @global_umax_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; --------------------------------------------------------------------------------
define amdgpu_ps float @global_umin_saddr_i32_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_umin_saddr_i32_rtn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umin v0, v0, v1, s[2:3] glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_umin_saddr_i32_rtn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umin v0, v0, v1, s[2:3] glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_umin_saddr_i32_rtn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umin v0, v0, v1, s[2:3] glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umin_saddr_i32_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u32 v0, v0, v1, s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umin_saddr_i32_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_min_u32 v0, v0, v1, s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2743,22 +2992,31 @@ define amdgpu_ps float @global_umin_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
}
define amdgpu_ps float @global_umin_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_umin_saddr_i32_rtn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umin v0, v0, v1, s[2:3] offset:-128 glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_umin_saddr_i32_rtn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umin v0, v0, v1, s[2:3] offset:-128 glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_umin_saddr_i32_rtn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umin v0, v0, v1, s[2:3] offset:-128 glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umin_saddr_i32_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u32 v0, v0, v1, s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umin_saddr_i32_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_min_u32 v0, v0, v1, s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2769,19 +3027,30 @@ define amdgpu_ps float @global_umin_saddr_i32_rtn_neg128(ptr addrspace(1) inreg
}
define amdgpu_ps void @global_umin_saddr_i32_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_umin_saddr_i32_nortn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umin v0, v1, s[2:3]
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_umin_saddr_i32_nortn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umin v0, v1, s[2:3]
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_umin_saddr_i32_nortn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umin v0, v1, s[2:3]
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_umin_saddr_i32_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u32 v0, v1, s[2:3]
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umin_saddr_i32_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u32 v0, v1, s[2:3]
+; GFX12-NEXT: global_atomic_min_u32 v0, v1, s[2:3] scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2790,19 +3059,30 @@ define amdgpu_ps void @global_umin_saddr_i32_nortn(ptr addrspace(1) inreg %sbase
}
define amdgpu_ps void @global_umin_saddr_i32_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i32 %data) {
-; GCN-LABEL: global_umin_saddr_i32_nortn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umin v0, v1, s[2:3] offset:-128
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_umin_saddr_i32_nortn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umin v0, v1, s[2:3] offset:-128
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_umin_saddr_i32_nortn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umin v0, v1, s[2:3] offset:-128
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_umin_saddr_i32_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u32 v0, v1, s[2:3] offset:-128
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umin_saddr_i32_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u32 v0, v1, s[2:3] offset:-128
+; GFX12-NEXT: global_atomic_min_u32 v0, v1, s[2:3] offset:-128 scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2812,22 +3092,31 @@ define amdgpu_ps void @global_umin_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
}
define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_umin_saddr_i64_rtn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umin_x2 v[0:1], v0, v[1:2], s[2:3] glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_umin_saddr_i64_rtn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umin_x2 v[0:1], v0, v[1:2], s[2:3] glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_umin_saddr_i64_rtn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umin_x2 v[0:1], v0, v[1:2], s[2:3] glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umin_saddr_i64_rtn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u64 v[0:1], v0, v[1:2], s[2:3] glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umin_saddr_i64_rtn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_min_u64 v[0:1], v0, v[1:2], s[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2837,22 +3126,31 @@ define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn(ptr addrspace(1) inreg %
}
define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_umin_saddr_i64_rtn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umin_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
-; GCN-NEXT: s_waitcnt vmcnt(0)
-; GCN-NEXT: ; return to shader part epilog
+; GFX9-LABEL: global_umin_saddr_i64_rtn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umin_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: ; return to shader part epilog
+;
+; GFX10-LABEL: global_umin_saddr_i64_rtn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umin_x2 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: ; return to shader part epilog
;
; GFX11-LABEL: global_umin_saddr_i64_rtn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u64 v[0:1], v0, v[1:2], s[2:3] offset:-128 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: global_umin_saddr_i64_rtn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN
+; GFX12-NEXT: global_atomic_min_u64 v[0:1], v0, v[1:2], s[2:3] offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_SE
; GFX12-NEXT: s_wait_loadcnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: ; return to shader part epilog
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2863,19 +3161,30 @@ define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn_neg128(ptr addrspace(1)
}
define amdgpu_ps void @global_umin_saddr_i64_nortn(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_umin_saddr_i64_nortn:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umin_x2 v0, v[1:2], s[2:3]
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_umin_saddr_i64_nortn:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umin_x2 v0, v[1:2], s[2:3]
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_umin_saddr_i64_nortn:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umin_x2 v0, v[1:2], s[2:3]
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_umin_saddr_i64_nortn:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u64 v0, v[1:2], s[2:3]
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umin_saddr_i64_nortn:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u64 v0, v[1:2], s[2:3]
+; GFX12-NEXT: global_atomic_min_u64 v0, v[1:2], s[2:3] scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
@@ -2884,19 +3193,30 @@ define amdgpu_ps void @global_umin_saddr_i64_nortn(ptr addrspace(1) inreg %sbase
}
define amdgpu_ps void @global_umin_saddr_i64_nortn_neg128(ptr addrspace(1) inreg %sbase, i32 %voffset, i64 %data) {
-; GCN-LABEL: global_umin_saddr_i64_nortn_neg128:
-; GCN: ; %bb.0:
-; GCN-NEXT: global_atomic_umin_x2 v0, v[1:2], s[2:3] offset:-128
-; GCN-NEXT: s_endpgm
+; GFX9-LABEL: global_umin_saddr_i64_nortn_neg128:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: global_atomic_umin_x2 v0, v[1:2], s[2:3] offset:-128
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: global_umin_saddr_i64_nortn_neg128:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: global_atomic_umin_x2 v0, v[1:2], s[2:3] offset:-128
+; GFX10-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX10-NEXT: buffer_gl0_inv
+; GFX10-NEXT: s_endpgm
;
; GFX11-LABEL: global_umin_saddr_i64_nortn_neg128:
; GFX11: ; %bb.0:
; GFX11-NEXT: global_atomic_min_u64 v0, v[1:2], s[2:3] offset:-128
+; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
+; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_endpgm
;
; GFX12-LABEL: global_umin_saddr_i64_nortn_neg128:
; GFX12: ; %bb.0:
-; GFX12-NEXT: global_atomic_min_u64 v0, v[1:2], s[2:3] offset:-128
+; GFX12-NEXT: global_atomic_min_u64 v0, v[1:2], s[2:3] offset:-128 scope:SCOPE_SE
+; GFX12-NEXT: s_wait_storecnt 0x0
+; GFX12-NEXT: global_inv scope:SCOPE_SE
; GFX12-NEXT: s_endpgm
%zext.offset = zext i32 %voffset to i64
%gep0 = getelementptr inbounds i8, ptr addrspace(1) %sbase, i64 %zext.offset
diff --git a/llvm/test/CodeGen/AMDGPU/implicitarg-offset-attributes.ll b/llvm/test/CodeGen/AMDGPU/implicitarg-offset-attributes.ll
index 86969396eaf2f8..868c3308d7c7a6 100644
--- a/llvm/test/CodeGen/AMDGPU/implicitarg-offset-attributes.ll
+++ b/llvm/test/CodeGen/AMDGPU/implicitarg-offset-attributes.ll
@@ -276,23 +276,23 @@ attributes #0 = { nocallback nofree nosync nounwind speculatable willreturn memo
;.
; V4: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
-; V4: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
-; V4: attributes #[[ATTR2]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
-; V4: attributes #[[ATTR3]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
-; V4: attributes #[[ATTR4]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
-; V4: attributes #[[ATTR5]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V4: attributes #[[ATTR1]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V4: attributes #[[ATTR2]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V4: attributes #[[ATTR3]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V4: attributes #[[ATTR4]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V4: attributes #[[ATTR5]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
;.
; V5: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
-; V5: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
-; V5: attributes #[[ATTR2]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
-; V5: attributes #[[ATTR3]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
-; V5: attributes #[[ATTR4]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V5: attributes #[[ATTR1]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V5: attributes #[[ATTR2]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V5: attributes #[[ATTR3]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V5: attributes #[[ATTR4]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
;.
; V6: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
-; V6: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
-; V6: attributes #[[ATTR2]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
-; V6: attributes #[[ATTR3]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
-; V6: attributes #[[ATTR4]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V6: attributes #[[ATTR1]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V6: attributes #[[ATTR2]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V6: attributes #[[ATTR3]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-default-queue" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
+; V6: attributes #[[ATTR4]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-wwm" }
;.
; V4: [[META0:![0-9]+]] = !{i32 1, !"amdhsa_code_object_version", i32 400}
;.
diff --git a/llvm/test/CodeGen/AMDGPU/indirect-call-set-from-other-function.ll b/llvm/test/CodeGen/AMDGPU/indirect-call-set-from-other-function.ll
index 727155c4b91002..eb123f83c9bc27 100644
--- a/llvm/test/CodeGen/AMDGPU/indirect-call-set-from-other-function.ll
+++ b/llvm/test/CodeGen/AMDGPU/indirect-call-set-from-other-function.ll
@@ -67,5 +67,5 @@ if.end:
ret void
}
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/issue120256-annotate-constexpr-addrspacecast.ll b/llvm/test/CodeGen/AMDGPU/issue120256-annotate-constexpr-addrspacecast.ll
index c537a0abd705e6..e214ad56df9b72 100644
--- a/llvm/test/CodeGen/AMDGPU/issue120256-annotate-constexpr-addrspacecast.ll
+++ b/llvm/test/CodeGen/AMDGPU/issue120256-annotate-constexpr-addrspacecast.ll
@@ -55,8 +55,8 @@ define amdgpu_kernel void @issue120256_private(ptr addrspace(1) %out) {
; FIXME: Inference of amdgpu-no-queue-ptr should not depend on code object version.
!0 = !{i32 1, !"amdhsa_code_object_version", i32 400}
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
; CHECK: [[META0:![0-9]+]] = !{i32 1, !"amdhsa_code_object_version", i32 400}
;.
diff --git a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma-async.ll b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma-async.ll
index 9d9ef15376c028..fb5321e575de2b 100644
--- a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma-async.ll
+++ b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma-async.ll
@@ -30,7 +30,7 @@ define amdgpu_kernel void @lds_wg_fence_release_single32(ptr addrspace(3) %lds)
ret void
}
-define amdgpu_kernel void @lds_async_dma_wg_fence_release_single32(ptr addrspace(1) %g, ptr addrspace(3) %lds) #0 {
+define amdgpu_kernel void @lds_async_dma_wg_fence_release_single32(ptr addrspace(1) %g, ptr addrspace(3) %lds) #1 {
; GFX1250-LABEL: lds_async_dma_wg_fence_release_single32:
; GFX1250: ; %bb.0:
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
@@ -62,4 +62,5 @@ define amdgpu_kernel void @lds_async_dma_wg_fence_release_single32(ptr addrspace
ret void
}
-attributes #0 = { nounwind "amdgpu-flat-work-group-size"="32,32" }
+attributes #0 = { nounwind "amdgpu-flat-work-group-size"="32,32" "amdgpu-no-async" }
+attributes #1 = { nounwind "amdgpu-flat-work-group-size"="32,32" }
diff --git a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma.ll b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma.ll
index 6472fde936a393..80cc670e8521de 100644
--- a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma.ll
+++ b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-lds-dma.ll
@@ -134,7 +134,7 @@ define amdgpu_kernel void @lds_wg_fence_release_single64(ptr addrspace(3) %lds)
ret void
}
-define amdgpu_kernel void @lds_dma_wg_fence_release_single32(ptr addrspace(8) %rsrc, ptr addrspace(3) %lds) #0 {
+define amdgpu_kernel void @lds_dma_wg_fence_release_single32(ptr addrspace(8) %rsrc, ptr addrspace(3) %lds) #2 {
; GFX9-LABEL: lds_dma_wg_fence_release_single32:
; GFX9: ; %bb.0:
; GFX9-NEXT: s_mov_b64 s[6:7], s[4:5]
@@ -246,5 +246,6 @@ define amdgpu_kernel void @lds_dma_wg_fence_release_single32(ptr addrspace(8) %r
ret void
}
-attributes #0 = { nounwind "amdgpu-flat-work-group-size"="32,32" }
-attributes #1 = { nounwind "amdgpu-flat-work-group-size"="64,64" }
+attributes #0 = { nounwind "amdgpu-flat-work-group-size"="32,32" "amdgpu-no-async" }
+attributes #1 = { nounwind "amdgpu-flat-work-group-size"="64,64" "amdgpu-no-async" }
+attributes #2 = { nounwind "amdgpu-flat-work-group-size"="32,32" }
diff --git a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-memops.ll b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-memops.ll
index 5c91713507e205..7128bab5b8e0c6 100644
--- a/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-memops.ll
+++ b/llvm/test/CodeGen/AMDGPU/memory-legalizer-single-wave-workgroup-memops.ll
@@ -3411,6 +3411,6 @@ define amdgpu_kernel void @flat_wg_st_seq_cst_multi(ptr addrspace(0) %p, i32 %x)
ret void
}
-attributes #0 = { nounwind "amdgpu-flat-work-group-size"="32,32" }
-attributes #1 = { nounwind "amdgpu-flat-work-group-size"="64,64" }
-attributes #2 = { nounwind "amdgpu-flat-work-group-size"="64,256" }
+attributes #0 = { nounwind "amdgpu-flat-work-group-size"="32,32" "amdgpu-no-async" }
+attributes #1 = { nounwind "amdgpu-flat-work-group-size"="64,64" "amdgpu-no-async" }
+attributes #2 = { nounwind "amdgpu-flat-work-group-size"="64,256" "amdgpu-no-async" }
diff --git a/llvm/test/CodeGen/AMDGPU/propagate-flat-work-group-size.ll b/llvm/test/CodeGen/AMDGPU/propagate-flat-work-group-size.ll
index 4a14d2ddda4cdc..89d36220ef134b 100644
--- a/llvm/test/CodeGen/AMDGPU/propagate-flat-work-group-size.ll
+++ b/llvm/test/CodeGen/AMDGPU/propagate-flat-work-group-size.ll
@@ -202,13 +202,13 @@ attributes #5 = { "amdgpu-flat-work-group-size"="128,512" }
attributes #6 = { "amdgpu-flat-work-group-size"="512,512" }
attributes #7 = { "amdgpu-flat-work-group-size"="64,256" }
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-flat-work-group-size"="1,256" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR1]] = { "amdgpu-flat-work-group-size"="64,128" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR2]] = { "amdgpu-flat-work-group-size"="128,512" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR3]] = { "amdgpu-flat-work-group-size"="64,64" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR4]] = { "amdgpu-flat-work-group-size"="128,256" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR5]] = { "amdgpu-flat-work-group-size"="512,1024" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR6]] = { "amdgpu-flat-work-group-size"="512,512" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR7]] = { "amdgpu-flat-work-group-size"="64,256" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR8]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-flat-work-group-size"="1,256" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR1]] = { "amdgpu-flat-work-group-size"="64,128" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR2]] = { "amdgpu-flat-work-group-size"="128,512" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR3]] = { "amdgpu-flat-work-group-size"="64,64" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR4]] = { "amdgpu-flat-work-group-size"="128,256" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR5]] = { "amdgpu-flat-work-group-size"="512,1024" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR6]] = { "amdgpu-flat-work-group-size"="512,512" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR7]] = { "amdgpu-flat-work-group-size"="64,256" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR8]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/propagate-waves-per-eu.ll b/llvm/test/CodeGen/AMDGPU/propagate-waves-per-eu.ll
index e69d9b182f6494..3887959c615615 100644
--- a/llvm/test/CodeGen/AMDGPU/propagate-waves-per-eu.ll
+++ b/llvm/test/CodeGen/AMDGPU/propagate-waves-per-eu.ll
@@ -399,25 +399,25 @@ attributes #17 = { "amdgpu-waves-per-eu"="5,8" }
attributes #18 = { "amdgpu-waves-per-eu"="9,10" }
attributes #19 = { "amdgpu-waves-per-eu"="8,9" }
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="2,8" }
-; CHECK: attributes #[[ATTR1]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="1,8" }
-; CHECK: attributes #[[ATTR2]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="1,2" }
-; CHECK: attributes #[[ATTR3]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="1,4" }
-; CHECK: attributes #[[ATTR4]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="9,9" }
-; CHECK: attributes #[[ATTR5]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="1,1" }
-; CHECK: attributes #[[ATTR6]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="9,10" }
-; CHECK: attributes #[[ATTR7]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="2,9" }
-; CHECK: attributes #[[ATTR8]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="3,8" }
-; CHECK: attributes #[[ATTR9]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR10]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR11]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="1,123" }
-; CHECK: attributes #[[ATTR12]] = { "amdgpu-flat-work-group-size"="1,512" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR13]] = { "amdgpu-flat-work-group-size"="1,512" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="3,6" }
-; CHECK: attributes #[[ATTR14]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="3,6" }
-; CHECK: attributes #[[ATTR15]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="4,8" }
-; CHECK: attributes #[[ATTR16]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="6,8" }
-; CHECK: attributes #[[ATTR17]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="5,5" }
-; CHECK: attributes #[[ATTR18]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="5,8" }
-; CHECK: attributes #[[ATTR19]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="9,10" }
-; CHECK: attributes #[[ATTR20]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="8,9" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="2,8" }
+; CHECK: attributes #[[ATTR1]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="1,8" }
+; CHECK: attributes #[[ATTR2]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="1,2" }
+; CHECK: attributes #[[ATTR3]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="1,4" }
+; CHECK: attributes #[[ATTR4]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="9,9" }
+; CHECK: attributes #[[ATTR5]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="1,1" }
+; CHECK: attributes #[[ATTR6]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="9,10" }
+; CHECK: attributes #[[ATTR7]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="2,9" }
+; CHECK: attributes #[[ATTR8]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="3,8" }
+; CHECK: attributes #[[ATTR9]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR10]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR11]] = { "amdgpu-flat-work-group-size"="1,64" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="1,123" }
+; CHECK: attributes #[[ATTR12]] = { "amdgpu-flat-work-group-size"="1,512" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR13]] = { "amdgpu-flat-work-group-size"="1,512" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="3,6" }
+; CHECK: attributes #[[ATTR14]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="3,6" }
+; CHECK: attributes #[[ATTR15]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="4,8" }
+; CHECK: attributes #[[ATTR16]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="6,8" }
+; CHECK: attributes #[[ATTR17]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="5,5" }
+; CHECK: attributes #[[ATTR18]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="5,8" }
+; CHECK: attributes #[[ATTR19]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="9,10" }
+; CHECK: attributes #[[ATTR20]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "amdgpu-waves-per-eu"="8,9" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/recursive_global_initializer.ll b/llvm/test/CodeGen/AMDGPU/recursive_global_initializer.ll
index b6820ea6bca3c2..31160e158c4d1e 100644
--- a/llvm/test/CodeGen/AMDGPU/recursive_global_initializer.ll
+++ b/llvm/test/CodeGen/AMDGPU/recursive_global_initializer.ll
@@ -19,5 +19,5 @@ define void @hoge() {
ret void
}
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/remove-no-kernel-id-attribute.ll b/llvm/test/CodeGen/AMDGPU/remove-no-kernel-id-attribute.ll
index a7c9a55eafa08d..6ea9faff7ad6d8 100644
--- a/llvm/test/CodeGen/AMDGPU/remove-no-kernel-id-attribute.ll
+++ b/llvm/test/CodeGen/AMDGPU/remove-no-kernel-id-attribute.ll
@@ -186,12 +186,12 @@ define amdgpu_kernel void @kernel_lds_recursion() {
!1 = !{i32 1, !"amdhsa_code_object_version", i32 400}
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR2]] = { "amdgpu-lds-size"="2" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR2]] = { "amdgpu-lds-size"="2" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
; CHECK: attributes #[[ATTR3]] = { "amdgpu-lds-size"="4" }
-; CHECK: attributes #[[ATTR4]] = { "amdgpu-lds-size"="2" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR5]] = { "amdgpu-lds-size"="4" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR4]] = { "amdgpu-lds-size"="2" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR5]] = { "amdgpu-lds-size"="4" "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
; CHECK: attributes #[[ATTR6:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(none) }
; CHECK: attributes #[[ATTR7:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/simple-indirect-call-2.ll b/llvm/test/CodeGen/AMDGPU/simple-indirect-call-2.ll
index 206c1947c8afc7..d6e18270337308 100644
--- a/llvm/test/CodeGen/AMDGPU/simple-indirect-call-2.ll
+++ b/llvm/test/CodeGen/AMDGPU/simple-indirect-call-2.ll
@@ -101,11 +101,11 @@ entry:
}
;.
-; NO: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; NO: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
-; OW: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; OW: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
-; CW: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CW: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
; NO: [[META0]] = !{ptr @bar1, ptr @bar2}
;.
diff --git a/llvm/test/CodeGen/AMDGPU/simple-indirect-call.ll b/llvm/test/CodeGen/AMDGPU/simple-indirect-call.ll
index 66f1b950659e9b..39aa64d1d98f97 100644
--- a/llvm/test/CodeGen/AMDGPU/simple-indirect-call.ll
+++ b/llvm/test/CodeGen/AMDGPU/simple-indirect-call.ll
@@ -57,7 +57,7 @@ define amdgpu_kernel void @test_simple_indirect_call() {
;.
-; ATTRIBUTOR_GCN: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; ATTRIBUTOR_GCN: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
; ATTRIBUTOR_GCN: [[META0]] = !{i32 1, i32 5, i32 6, i32 16}
;.
diff --git a/llvm/test/CodeGen/AMDGPU/uniform-work-group-attribute-missing.ll b/llvm/test/CodeGen/AMDGPU/uniform-work-group-attribute-missing.ll
index b716af2d25cbf1..77a1d13fcbdf1b 100644
--- a/llvm/test/CodeGen/AMDGPU/uniform-work-group-attribute-missing.ll
+++ b/llvm/test/CodeGen/AMDGPU/uniform-work-group-attribute-missing.ll
@@ -31,6 +31,6 @@ define amdgpu_kernel void @kernel1() #1 {
attributes #0 = { "uniform-work-group-size" }
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "uniform-work-group-size" }
-; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "uniform-work-group-size" }
+; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/uniform-work-group-multistep.ll b/llvm/test/CodeGen/AMDGPU/uniform-work-group-multistep.ll
index 870c60f771060f..78754a10944091 100644
--- a/llvm/test/CodeGen/AMDGPU/uniform-work-group-multistep.ll
+++ b/llvm/test/CodeGen/AMDGPU/uniform-work-group-multistep.ll
@@ -96,7 +96,7 @@ define amdgpu_kernel void @kernel2() #0 {
attributes #0 = { "uniform-work-group-size" }
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
; CHECK: attributes #[[ATTR1]] = { "uniform-work-group-size" }
-; CHECK: attributes #[[ATTR2]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "uniform-work-group-size" }
+; CHECK: attributes #[[ATTR2]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "uniform-work-group-size" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/uniform-work-group-nested-function-calls.ll b/llvm/test/CodeGen/AMDGPU/uniform-work-group-nested-function-calls.ll
index 17263b622f92eb..0a3f12dcad67f7 100644
--- a/llvm/test/CodeGen/AMDGPU/uniform-work-group-nested-function-calls.ll
+++ b/llvm/test/CodeGen/AMDGPU/uniform-work-group-nested-function-calls.ll
@@ -41,6 +41,6 @@ define amdgpu_kernel void @kernel3() #2 {
attributes #2 = { "uniform-work-group-size" }
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "uniform-work-group-size" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "uniform-work-group-size" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/uniform-work-group-prevent-attribute-propagation.ll b/llvm/test/CodeGen/AMDGPU/uniform-work-group-prevent-attribute-propagation.ll
index 110366f1a4da45..c2b458502c9444 100644
--- a/llvm/test/CodeGen/AMDGPU/uniform-work-group-prevent-attribute-propagation.ll
+++ b/llvm/test/CodeGen/AMDGPU/uniform-work-group-prevent-attribute-propagation.ll
@@ -41,6 +41,6 @@ define amdgpu_kernel void @kernel2() #2 {
attributes #1 = { "uniform-work-group-size" }
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "uniform-work-group-size" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "uniform-work-group-size" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/uniform-work-group-propagate-attribute.ll b/llvm/test/CodeGen/AMDGPU/uniform-work-group-propagate-attribute.ll
index ebea26ef56c687..9b9ee4f3402d14 100644
--- a/llvm/test/CodeGen/AMDGPU/uniform-work-group-propagate-attribute.ll
+++ b/llvm/test/CodeGen/AMDGPU/uniform-work-group-propagate-attribute.ll
@@ -51,8 +51,8 @@ define amdgpu_kernel void @kernel2() #1 {
attributes #0 = { nounwind }
attributes #1 = { "uniform-work-group-size" }
;.
-; CHECK: attributes #[[ATTR0]] = { nounwind "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR0]] = { nounwind "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR1]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
; CHECK: attributes #[[ATTR2]] = { nounwind }
; CHECK: attributes #[[ATTR3]] = { "uniform-work-group-size" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/uniform-work-group-recursion-test.ll b/llvm/test/CodeGen/AMDGPU/uniform-work-group-recursion-test.ll
index b051bb577ad778..acf2e150ceed5c 100644
--- a/llvm/test/CodeGen/AMDGPU/uniform-work-group-recursion-test.ll
+++ b/llvm/test/CodeGen/AMDGPU/uniform-work-group-recursion-test.ll
@@ -101,7 +101,7 @@ define amdgpu_kernel void @kernel(ptr addrspace(1) %m) #1 {
attributes #0 = { nounwind readnone }
attributes #1 = { "uniform-work-group-size" }
;.
-; CHECK: attributes #[[ATTR0]] = { nounwind memory(none) "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
-; CHECK: attributes #[[ATTR1]] = { nounwind memory(none) "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "uniform-work-group-size" }
-; CHECK: attributes #[[ATTR2]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "uniform-work-group-size" }
+; CHECK: attributes #[[ATTR0]] = { nounwind memory(none) "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR1]] = { nounwind memory(none) "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "uniform-work-group-size" }
+; CHECK: attributes #[[ATTR2]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" "uniform-work-group-size" }
;.
diff --git a/llvm/test/CodeGen/AMDGPU/uniform-work-group-test.ll b/llvm/test/CodeGen/AMDGPU/uniform-work-group-test.ll
index e44e37b3ef31b1..deae4d794f1747 100644
--- a/llvm/test/CodeGen/AMDGPU/uniform-work-group-test.ll
+++ b/llvm/test/CodeGen/AMDGPU/uniform-work-group-test.ll
@@ -60,5 +60,5 @@ define amdgpu_kernel void @kernel3() {
}
;.
-; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
+; CHECK: attributes #[[ATTR0]] = { "amdgpu-no-async" "amdgpu-no-cluster-id-x" "amdgpu-no-cluster-id-y" "amdgpu-no-cluster-id-z" "amdgpu-no-completion-action" "amdgpu-no-default-queue" "amdgpu-no-dispatch-id" "amdgpu-no-dispatch-ptr" "amdgpu-no-flat-scratch-init" "amdgpu-no-heap-ptr" "amdgpu-no-hostcall-ptr" "amdgpu-no-implicitarg-ptr" "amdgpu-no-lds-kernel-id" "amdgpu-no-multigrid-sync-arg" "amdgpu-no-queue-ptr" "amdgpu-no-workgroup-id-x" "amdgpu-no-workgroup-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" "amdgpu-no-wwm" }
;.
More information about the llvm-branch-commits
mailing list