[llvm] [AMDGPU] Don't assume a hazard for empty inline asm (PR #223526)
Lixun Zhang via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 14 14:15:41 PDT 2026
https://github.com/zhanglx13 updated https://github.com/llvm/llvm-project/pull/223526
>From 7acab9817467639479753c5ec59d0703a3edb63f Mon Sep 17 00:00:00 2001
From: Lixun Zhang <lixun.zhang at amd.com>
Date: Mon, 14 Sep 2026 21:01:13 +0000
Subject: [PATCH] [AMDGPU] Don't assume a hazard for empty inline asm
GCNHazardRecognizer assumes that any inline asm may have a dst-sel
forwarding hazard. A VALU that reads a register defined by inline asm
therefore gets a wait state, and so does inline asm that reads the result
of a real dst-sel forwarding producer.
An inline asm with an empty asm string emits no instruction. Frontends use
one as a register-class constraint on a tied operand: Triton pins MFMA
accumulators to AGPRs with an empty "=a,0" inline asm next to each MFMA.
Such an inline asm can neither produce nor consume a hazard. Skip it in
checkVALUHazards and checkInlineAsmHazards. A hazard carried by the value
is still found at the instruction that produced it, because wait-state
counting already walks past inline asm. It is then padded once, in front
of the real reader, instead of once on each side of the pin.
Inline asm with any content, including a comment only, keeps the
conservative assumption.
On a Triton GEMM kernel with every MFMA accumulator pinned, this removes
all 90 s_nop from the hot loop.
---
.../lib/Target/AMDGPU/GCNHazardRecognizer.cpp | 21 ++++-
.../AMDGPU/inlineasm-empty-no-hazard-mfma.ll | 64 ++++++++++++++
.../AMDGPU/inlineasm-empty-no-hazard.ll | 87 +++++++++++++++++++
3 files changed, 170 insertions(+), 2 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/inlineasm-empty-no-hazard-mfma.ll
create mode 100644 llvm/test/CodeGen/AMDGPU/inlineasm-empty-no-hazard.ll
diff --git a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
index 33c367bbcd7484..67112448a1fa6a 100644
--- a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
@@ -1452,6 +1452,19 @@ int GCNHazardRecognizer::checkVALUHazardsHelper(
return checkUniformWindowVALUHazardsHelper(Def.getReg());
}
+/// An inline asm with an empty asm string emits no instruction. Frontends use
+/// one as a pure register-class constraint on a tied operand (e.g. a "=a,0"
+/// constraint that keeps an MFMA accumulator in AGPRs), so it cannot produce
+/// a hazard of its own. Any hazard carried by the value is still found at the
+/// instruction that produced it, since wait-state counting already looks past
+/// inline asm.
+static bool isEmptyInlineAsm(const MachineInstr &MI) {
+ if (!MI.isInlineAsm())
+ return false;
+ const char *Asm = MI.getOperand(InlineAsm::MIOp_AsmString).getSymbolName();
+ return !Asm || *Asm == '\0';
+}
+
/// Dest sel forwarding issue occurs if additional logic is needed to swizzle /
/// pack the computed value into correct bit position of the dest register. This
/// occurs if we have SDWA with dst_sel != DWORD or if we have op_sel with
@@ -1564,7 +1577,7 @@ int GCNHazardRecognizer::checkVALUHazards(MachineInstr *VALU) const {
return consumesDstSelForwardingOperand(VALU, ForwardedDst, TRI);
}
- if (ProducerMI.isInlineAsm()) {
+ if (ProducerMI.isInlineAsm() && !isEmptyInlineAsm(ProducerMI)) {
// Assume inline asm has dst forwarding hazard
for (auto &Def : ProducerMI.all_defs()) {
if (consumesDstSelForwardingOperand(VALU, &Def, TRI))
@@ -1668,6 +1681,10 @@ int GCNHazardRecognizer::checkInlineAsmHazards(MachineInstr *IA) const {
!ST.hasCvtScaleForwardingHazard())
return 0;
+ // An empty inline asm executes nothing, so it cannot consume a hazard.
+ if (isEmptyInlineAsm(*IA))
+ return 0;
+
const MachineRegisterInfo &MRI = MF.getRegInfo();
int WaitStatesNeeded = 0;
@@ -1694,7 +1711,7 @@ int GCNHazardRecognizer::checkInlineAsmHazards(MachineInstr *IA) const {
return IA->modifiesRegister(Dst->getReg(), &TRI) ||
IA->readsRegister(Dst->getReg(), &TRI);
- if (ProducerMI.isInlineAsm()) {
+ if (ProducerMI.isInlineAsm() && !isEmptyInlineAsm(ProducerMI)) {
// If MI is inline asm, assume it has dst forwarding hazard
for (auto &Def : ProducerMI.all_defs()) {
if (IA->modifiesRegister(Def.getReg(), &TRI) ||
diff --git a/llvm/test/CodeGen/AMDGPU/inlineasm-empty-no-hazard-mfma.ll b/llvm/test/CodeGen/AMDGPU/inlineasm-empty-no-hazard-mfma.ll
new file mode 100644
index 00000000000000..7254576b500ae0
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/inlineasm-empty-no-hazard-mfma.ll
@@ -0,0 +1,64 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=amdgpu9.50 -O3 --enable-misched=0 --enable-post-misched=0 -o - %s | FileCheck %s
+
+; An empty "=a,0" inline asm keeps an MFMA accumulator in AGPRs without emitting
+; anything, so the MFMA that reads it needs no wait state. A non-empty inline asm
+; still gets one.
+
+declare <4 x float> @llvm.amdgcn.mfma.f32.16x16x32.f16(<8 x half>, <8 x half>, <4 x float>, i32, i32, i32)
+
+define amdgpu_kernel void @empty_pin(ptr addrspace(3) %a, ptr addrspace(3) %b, ptr addrspace(3) %out) {
+; CHECK-LABEL: empty_pin:
+; CHECK: ; %bb.0:
+; CHECK-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; CHECK-NEXT: s_waitcnt lgkmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, s0
+; CHECK-NEXT: ds_read_b128 v[4:7], v0
+; CHECK-NEXT: v_mov_b32_e32 v0, s1
+; CHECK-NEXT: ds_read_b128 v[8:11], v0
+; CHECK-NEXT: s_waitcnt lgkmcnt(0)
+; CHECK-NEXT: v_mfma_f32_16x16x32_f16 a[0:3], v[4:7], v[8:11], 0
+; CHECK-NEXT: ;;#ASMSTART
+; CHECK-NEXT: ;;#ASMEND
+; CHECK-NEXT: v_mfma_f32_16x16x32_f16 a[0:3], v[4:7], v[8:11], a[0:3]
+; CHECK-NEXT: v_mov_b32_e32 v4, s2
+; CHECK-NEXT: s_nop 6
+; CHECK-NEXT: ds_write_b128 v4, a[0:3]
+; CHECK-NEXT: s_endpgm
+ %va = load <8 x half>, ptr addrspace(3) %a, align 16
+ %vb = load <8 x half>, ptr addrspace(3) %b, align 16
+ %m0 = tail call <4 x float> @llvm.amdgcn.mfma.f32.16x16x32.f16(<8 x half> %va, <8 x half> %vb, <4 x float> zeroinitializer, i32 0, i32 0, i32 0)
+ %pin = tail call <4 x float> asm "", "=a,0"(<4 x float> %m0)
+ %m1 = tail call <4 x float> @llvm.amdgcn.mfma.f32.16x16x32.f16(<8 x half> %va, <8 x half> %vb, <4 x float> %pin, i32 0, i32 0, i32 0)
+ store <4 x float> %m1, ptr addrspace(3) %out, align 16
+ ret void
+}
+
+define amdgpu_kernel void @comment_only_asm(ptr addrspace(3) %a, ptr addrspace(3) %b, ptr addrspace(3) %out) {
+; CHECK-LABEL: comment_only_asm:
+; CHECK: ; %bb.0:
+; CHECK-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; CHECK-NEXT: s_waitcnt lgkmcnt(0)
+; CHECK-NEXT: v_mov_b32_e32 v0, s0
+; CHECK-NEXT: ds_read_b128 v[4:7], v0
+; CHECK-NEXT: v_mov_b32_e32 v0, s1
+; CHECK-NEXT: ds_read_b128 v[8:11], v0
+; CHECK-NEXT: s_waitcnt lgkmcnt(0)
+; CHECK-NEXT: v_mfma_f32_16x16x32_f16 a[0:3], v[4:7], v[8:11], 0
+; CHECK-NEXT: ;;#ASMSTART
+; CHECK-NEXT: ; def a[0:3]
+; CHECK-NEXT: ;;#ASMEND
+; CHECK-NEXT: s_nop 0
+; CHECK-NEXT: v_mfma_f32_16x16x32_f16 a[0:3], v[4:7], v[8:11], a[0:3]
+; CHECK-NEXT: v_mov_b32_e32 v4, s2
+; CHECK-NEXT: s_nop 6
+; CHECK-NEXT: ds_write_b128 v4, a[0:3]
+; CHECK-NEXT: s_endpgm
+ %va = load <8 x half>, ptr addrspace(3) %a, align 16
+ %vb = load <8 x half>, ptr addrspace(3) %b, align 16
+ %m0 = tail call <4 x float> @llvm.amdgcn.mfma.f32.16x16x32.f16(<8 x half> %va, <8 x half> %vb, <4 x float> zeroinitializer, i32 0, i32 0, i32 0)
+ %pin = tail call <4 x float> asm "; def $0", "=a,0"(<4 x float> %m0)
+ %m1 = tail call <4 x float> @llvm.amdgcn.mfma.f32.16x16x32.f16(<8 x half> %va, <8 x half> %vb, <4 x float> %pin, i32 0, i32 0, i32 0)
+ store <4 x float> %m1, ptr addrspace(3) %out, align 16
+ ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/inlineasm-empty-no-hazard.ll b/llvm/test/CodeGen/AMDGPU/inlineasm-empty-no-hazard.ll
new file mode 100644
index 00000000000000..bd95f0ddf9a329
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/inlineasm-empty-no-hazard.ll
@@ -0,0 +1,87 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=amdgpu9.42 -o - %s | FileCheck %s
+; RUN: llc -mtriple=amdgpu9.50 -o - %s | FileCheck %s
+
+; An inline asm with an empty asm string emits no instruction; frontends use one
+; as a register-class constraint on a tied operand. The hazard recognizer must
+; not assume it has a dst-sel forwarding hazard, while a real hazard from the
+; instruction that produced the value is still padded, once. Inline asm that
+; contains anything, even only a comment, keeps the conservative assumption.
+
+declare i32 @llvm.amdgcn.cvt.scalef32.sr.fp8.f32(i32 %old, float %src, i32 %seed, float %scale, i32 %dst_sel)
+
+; Empty pin between two plain VALUs: no hazard.
+define amdgpu_ps void @empty_pin(ptr addrspace(1) %out, i32 %x, i32 %y, i32 %z) {
+; CHECK-LABEL: empty_pin:
+; CHECK: ; %bb.0:
+; CHECK-NEXT: v_add_u32_e32 v2, v2, v3
+; CHECK-NEXT: ;;#ASMSTART
+; CHECK-NEXT: ;;#ASMEND
+; CHECK-NEXT: v_add_u32_e32 v2, v2, v4
+; CHECK-NEXT: global_store_dword v[0:1], v2, off
+; CHECK-NEXT: s_endpgm
+ %a = add i32 %x, %y
+ %pin = call i32 asm "", "=v,0"(i32 %a)
+ %b = add i32 %pin, %z
+ store i32 %b, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; A real dst-sel forwarding producer before the pin still pads its reader, once.
+define amdgpu_ps void @real_hazard_through_empty_pin(ptr addrspace(1) %out, float %src, i32 %seed, float %scale, i32 %x) {
+; CHECK-LABEL: real_hazard_through_empty_pin:
+; CHECK: ; %bb.0:
+; CHECK-NEXT: global_load_dword v6, v[0:1], off
+; CHECK-NEXT: s_waitcnt vmcnt(0)
+; CHECK-NEXT: v_cvt_scalef32_sr_fp8_f32 v6, v2, v3, v4 op_sel:[0,0,0,1]
+; CHECK-NEXT: ;;#ASMSTART
+; CHECK-NEXT: ;;#ASMEND
+; CHECK-NEXT: s_nop 0
+; CHECK-NEXT: v_add_u32_e32 v2, v6, v5
+; CHECK-NEXT: global_store_dword v[0:1], v2, off
+; CHECK-NEXT: s_endpgm
+ %old = load i32, ptr addrspace(1) %out, align 4
+ %cvt = call i32 @llvm.amdgcn.cvt.scalef32.sr.fp8.f32(i32 %old, float %src, i32 %seed, float %scale, i32 2)
+ %pin = call i32 asm "", "=v,0"(i32 %cvt)
+ %sum = add i32 %pin, %x
+ store i32 %sum, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; An inline asm that emits an instruction keeps the conservative assumption.
+define amdgpu_ps void @nonempty_asm(ptr addrspace(1) %out, i32 %x, i32 %y, i32 %z) {
+; CHECK-LABEL: nonempty_asm:
+; CHECK: ; %bb.0:
+; CHECK-NEXT: v_add_u32_e32 v2, v2, v3
+; CHECK-NEXT: ;;#ASMSTART
+; CHECK-NEXT: v_mov_b32_e32 v2, v2
+; CHECK-NEXT: ;;#ASMEND
+; CHECK-NEXT: s_nop 0
+; CHECK-NEXT: v_add_u32_e32 v2, v2, v4
+; CHECK-NEXT: global_store_dword v[0:1], v2, off
+; CHECK-NEXT: s_endpgm
+ %a = add i32 %x, %y
+ %pin = call i32 asm "v_mov_b32_e32 $0, $1", "=v,0"(i32 %a)
+ %b = add i32 %pin, %z
+ store i32 %b, ptr addrspace(1) %out, align 4
+ ret void
+}
+
+; So does a comment-only inline asm: it is not treated as empty.
+define amdgpu_ps void @comment_only_asm(ptr addrspace(1) %out, i32 %x, i32 %y, i32 %z) {
+; CHECK-LABEL: comment_only_asm:
+; CHECK: ; %bb.0:
+; CHECK-NEXT: v_add_u32_e32 v2, v2, v3
+; CHECK-NEXT: ;;#ASMSTART
+; CHECK-NEXT: ; def v2
+; CHECK-NEXT: ;;#ASMEND
+; CHECK-NEXT: s_nop 0
+; CHECK-NEXT: v_add_u32_e32 v2, v2, v4
+; CHECK-NEXT: global_store_dword v[0:1], v2, off
+; CHECK-NEXT: s_endpgm
+ %a = add i32 %x, %y
+ %pin = call i32 asm "; def $0", "=v,0"(i32 %a)
+ %b = add i32 %pin, %z
+ store i32 %b, ptr addrspace(1) %out, align 4
+ ret void
+}
More information about the llvm-commits
mailing list