[llvm] [AMDGPU] Account for inline asm size in inst_pref_size calculation (PR #192306)
Adel Ejjeh via llvm-commits
llvm-commits at lists.llvm.org
Wed Apr 29 08:59:42 PDT 2026
https://github.com/adelejjeh updated https://github.com/llvm/llvm-project/pull/192306
>From b4784a19f5bcdcb730c9fb0b5333f845067e3ee9 Mon Sep 17 00:00:00 2001
From: Adel Ejjeh <adel.ejjeh at amd.com>
Date: Wed, 15 Apr 2026 13:25:36 -0500
Subject: [PATCH 01/11] [AMDGPU] Account for inline asm size in inst_pref_size
calculation
SIProgramInfo::getFunctionCodeSize() with IsLowerBound=true was
completely skipping inline assembly instructions, treating them as
zero bytes. This caused amdhsa_inst_pref_size to be severely
underestimated for kernels containing inline asm, defeating
instruction prefetch on gfx11+.
Use MCExpr label subtraction (.Lfunc_end - func_sym) to compute
exact function code size, resolved at assembly time. This avoids
inline asm string parsing which cannot reliably estimate code size
and risks overestimation (which causes prefetch of unmapped memory
and a fatal segfault).
Add a new AMDGPUMCExpr variant (AGVK_InstPrefSize) to compute
min(divideCeil(codeSize, 128), maxFieldVal) as a custom MCExpr,
following the same pattern as AGVK_Occupancy and AGVK_AlignTo.
Compute inst_pref_size in AMDGPUAsmPrinter::endFunction() where
.Lfunc_end has already been emitted in the correct position (after
the fix in #191526), and set the bits in ComputePGMRSrc3 before
emitting the kernel descriptor.
Remove the IsLowerBound parameter from getFunctionCodeSize() as it
is no longer needed for inst_pref_size calculation.
Co-Authored-By: Claude Opus 4 (1M context) <noreply at anthropic.com>
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 85 ++++++++++---------
.../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 38 +++++++++
.../Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h | 8 ++
llvm/lib/Target/AMDGPU/SIProgramInfo.cpp | 17 +---
llvm/lib/Target/AMDGPU/SIProgramInfo.h | 5 +-
.../test/CodeGen/AMDGPU/inst-prefetch-hint.ll | 36 ++++++--
.../AMDGPU/inst-prefetch-inline-asm.ll | 83 ++++++++++++++++++
7 files changed, 208 insertions(+), 64 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 94ff6c2daf330..c57e592afc2ad 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -233,6 +233,18 @@ void AMDGPUAsmPrinter::emitFunctionBodyStart() {
HSAMetadataStream->emitKernel(*MF, CurrentProgramInfo);
}
+/// Set bits in a kernel descriptor MCExpr field:
+/// return ((Dst & ~Mask) | (Value << Shift))
+static const MCExpr *setBits(const MCExpr *Dst, const MCExpr *Value,
+ uint32_t Mask, uint32_t Shift, MCContext &Ctx) {
+ const auto *Shft = MCConstantExpr::create(Shift, Ctx);
+ const auto *Msk = MCConstantExpr::create(Mask, Ctx);
+ Dst = MCBinaryExpr::createAnd(Dst, MCUnaryExpr::createNot(Msk, Ctx), Ctx);
+ Dst = MCBinaryExpr::createOr(Dst, MCBinaryExpr::createShl(Value, Shft, Ctx),
+ Ctx);
+ return Dst;
+}
+
void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
const SIMachineFunctionInfo &MFI = *MF->getInfo<SIMachineFunctionInfo>();
if (!MFI.isEntryFunction())
@@ -240,6 +252,34 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
assert(TM.getTargetTriple().getOS() == Triple::AMDHSA);
+ const GCNSubtarget &STM = MF->getSubtarget<GCNSubtarget>();
+ MCContext &Ctx = MF->getContext();
+
+ // Compute inst_pref_size using MCExpr label subtraction for exact code
+ // size. (Lfunc_end - func_sym) gives the exact function code size in bytes.
+ if (isGFX11Plus(STM)) {
+ const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
+ MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
+ MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
+
+ uint32_t Field, Shift, Width;
+ if (isGFX11(STM)) {
+ Field = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE;
+ Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT;
+ Width = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_WIDTH;
+ } else {
+ Field = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE;
+ Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT;
+ Width = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_WIDTH;
+ }
+ const MCExpr *InstPrefSizeExpr =
+ AMDGPUMCExpr::createInstPrefSize(CodeSizeExpr, Width, Ctx);
+
+ CurrentProgramInfo.ComputePGMRSrc3 =
+ setBits(CurrentProgramInfo.ComputePGMRSrc3, InstPrefSizeExpr, Field,
+ Shift, Ctx);
+ }
+
auto &Streamer = getTargetStreamer()->getStreamer();
auto &Context = Streamer.getContext();
auto &ObjectFileInfo = *Context.getObjectFileInfo();
@@ -253,8 +293,6 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
Streamer.emitValueToAlignment(Align(64), 0, 1, 0);
ReadOnlySection.ensureMinAlignment(Align(64));
- const GCNSubtarget &STM = MF->getSubtarget<GCNSubtarget>();
-
SmallString<128> KernelName;
getNameWithPrefix(KernelName, &MF->getFunction());
getTargetStreamer()->EmitAmdhsaKernelDescriptor(
@@ -1282,33 +1320,22 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
ProgInfo.LdsSize = STM.isAmdHsaOS() ? 0 : ProgInfo.LDSBlocks;
ProgInfo.EXCPEnable = 0;
- // return ((Dst & ~Mask) | (Value << Shift))
- auto SetBits = [&Ctx](const MCExpr *Dst, const MCExpr *Value, uint32_t Mask,
- uint32_t Shift) {
- const auto *Shft = MCConstantExpr::create(Shift, Ctx);
- const auto *Msk = MCConstantExpr::create(Mask, Ctx);
- Dst = MCBinaryExpr::createAnd(Dst, MCUnaryExpr::createNot(Msk, Ctx), Ctx);
- Dst = MCBinaryExpr::createOr(Dst, MCBinaryExpr::createShl(Value, Shft, Ctx),
- Ctx);
- return Dst;
- };
-
if (STM.hasGFX90AInsts()) {
ProgInfo.ComputePGMRSrc3 =
- SetBits(ProgInfo.ComputePGMRSrc3, ProgInfo.AccumOffset,
+ setBits(ProgInfo.ComputePGMRSrc3, ProgInfo.AccumOffset,
amdhsa::COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET,
- amdhsa::COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET_SHIFT);
+ amdhsa::COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET_SHIFT, Ctx);
ProgInfo.ComputePGMRSrc3 =
- SetBits(ProgInfo.ComputePGMRSrc3, CreateExpr(ProgInfo.TgSplit),
+ setBits(ProgInfo.ComputePGMRSrc3, CreateExpr(ProgInfo.TgSplit),
amdhsa::COMPUTE_PGM_RSRC3_GFX90A_TG_SPLIT,
- amdhsa::COMPUTE_PGM_RSRC3_GFX90A_TG_SPLIT_SHIFT);
+ amdhsa::COMPUTE_PGM_RSRC3_GFX90A_TG_SPLIT_SHIFT, Ctx);
}
if (STM.hasGFX1250Insts())
ProgInfo.ComputePGMRSrc3 =
- SetBits(ProgInfo.ComputePGMRSrc3, ProgInfo.NamedBarCnt,
+ setBits(ProgInfo.ComputePGMRSrc3, ProgInfo.NamedBarCnt,
amdhsa::COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT,
- amdhsa::COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT_SHIFT);
+ amdhsa::COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT_SHIFT, Ctx);
ProgInfo.Occupancy = createOccupancy(
STM.computeOccupancy(F, ProgInfo.LDSSize).second,
@@ -1327,26 +1354,6 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
", final occupancy is " + Twine(Occupancy));
F.getContext().diagnose(Diag);
}
-
- if (isGFX11Plus(STM)) {
- uint32_t CodeSizeInBytes = (uint32_t)std::min(
- ProgInfo.getFunctionCodeSize(MF, true /* IsLowerBound */),
- (uint64_t)std::numeric_limits<uint32_t>::max());
- uint32_t CodeSizeInLines = divideCeil(CodeSizeInBytes, 128);
- uint32_t Field, Shift, Width;
- if (isGFX11(STM)) {
- Field = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE;
- Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT;
- Width = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_WIDTH;
- } else {
- Field = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE;
- Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT;
- Width = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_WIDTH;
- }
- uint64_t InstPrefSize = std::min(CodeSizeInLines, (1u << Width) - 1);
- ProgInfo.ComputePGMRSrc3 = SetBits(ProgInfo.ComputePGMRSrc3,
- CreateExpr(InstPrefSize), Field, Shift);
- }
}
static unsigned getRsrcReg(CallingConv::ID CallConv) {
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index fd0a2a6a77d7e..48c4491f22db3 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -15,6 +15,7 @@
#include "llvm/MC/MCSymbol.h"
#include "llvm/MC/MCValue.h"
#include "llvm/Support/KnownBits.h"
+#include "llvm/Support/MathExtras.h"
#include "llvm/Support/raw_ostream.h"
#include <functional>
#include <optional>
@@ -74,6 +75,9 @@ void AMDGPUMCExpr::printImpl(raw_ostream &OS, const MCAsmInfo *MAI) const {
case AGVK_Occupancy:
OS << "occupancy(";
break;
+ case AGVK_InstPrefSize:
+ OS << "instprefsize(";
+ break;
case AGVK_Lit:
OS << "lit(";
break;
@@ -182,6 +186,30 @@ bool AMDGPUMCExpr::evaluateOccupancy(MCValue &Res,
return true;
}
+bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
+ const MCAssembler *Asm) const {
+ auto TryGetMCExprValue = [&](const MCExpr *Arg, uint64_t &ConstantValue) {
+ MCValue MCVal;
+ if (!Arg->evaluateAsRelocatable(MCVal, Asm) || !MCVal.isAbsolute())
+ return false;
+
+ ConstantValue = MCVal.getConstant();
+ return true;
+ };
+
+ assert(Args.size() == 2 &&
+ "AMDGPUMCExpr Argument count incorrect for InstPrefSize");
+ uint64_t CodeSizeInBytes = 0, FieldWidth = 0;
+ if (!TryGetMCExprValue(Args[0], CodeSizeInBytes) ||
+ !TryGetMCExprValue(Args[1], FieldWidth))
+ return false;
+
+ uint64_t CodeSizeInLines = divideCeil(CodeSizeInBytes, (uint64_t)128);
+ uint64_t MaxVal = (1u << FieldWidth) - 1;
+ Res = MCValue::get(std::min(CodeSizeInLines, MaxVal));
+ return true;
+}
+
bool AMDGPUMCExpr::isSymbolUsedInExpression(const MCSymbol *Sym,
const MCExpr *E) {
switch (E->getKind()) {
@@ -227,6 +255,8 @@ bool AMDGPUMCExpr::evaluateAsRelocatableImpl(MCValue &Res,
return evaluateTotalNumVGPR(Res, Asm);
case AGVK_Occupancy:
return evaluateOccupancy(Res, Asm);
+ case AGVK_InstPrefSize:
+ return evaluateInstPrefSize(Res, Asm);
case AGVK_Lit:
case AGVK_Lit64:
return Args[0]->evaluateAsRelocatable(Res, Asm);
@@ -279,6 +309,13 @@ const AMDGPUMCExpr *AMDGPUMCExpr::createTotalNumVGPR(const MCExpr *NumAGPR,
return create(AGVK_TotalNumVGPRs, {NumAGPR, NumVGPR}, Ctx);
}
+const AMDGPUMCExpr *
+AMDGPUMCExpr::createInstPrefSize(const MCExpr *CodeSizeBytes,
+ unsigned FieldWidth, MCContext &Ctx) {
+ return create(AGVK_InstPrefSize,
+ {CodeSizeBytes, MCConstantExpr::create(FieldWidth, Ctx)}, Ctx);
+}
+
const AMDGPUMCExpr *AMDGPUMCExpr::createLit(LitModifier Lit, int64_t Value,
MCContext &Ctx) {
assert(Lit == LitModifier::Lit || Lit == LitModifier::Lit64);
@@ -469,6 +506,7 @@ static void targetOpKnownBitsMapHelper(const MCExpr *Expr, KnownBitsMap &KBM,
case AMDGPUMCExpr::VariantKind::AGVK_TotalNumVGPRs:
case AMDGPUMCExpr::VariantKind::AGVK_AlignTo:
case AMDGPUMCExpr::VariantKind::AGVK_Occupancy:
+ case AMDGPUMCExpr::VariantKind::AGVK_InstPrefSize:
case AMDGPUMCExpr::VariantKind::AGVK_Lit:
case AMDGPUMCExpr::VariantKind::AGVK_Lit64: {
int64_t Val;
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index 96bd8f4cf3c13..52c984ac450a7 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -38,6 +38,7 @@ class AMDGPUMCExpr : public MCTargetExpr {
AGVK_TotalNumVGPRs,
AGVK_AlignTo,
AGVK_Occupancy,
+ AGVK_InstPrefSize,
AGVK_Lit,
AGVK_Lit64,
};
@@ -69,6 +70,7 @@ class AMDGPUMCExpr : public MCTargetExpr {
bool evaluateTotalNumVGPR(MCValue &Res, const MCAssembler *Asm) const;
bool evaluateAlignTo(MCValue &Res, const MCAssembler *Asm) const;
bool evaluateOccupancy(MCValue &Res, const MCAssembler *Asm) const;
+ bool evaluateInstPrefSize(MCValue &Res, const MCAssembler *Asm) const;
public:
static const AMDGPUMCExpr *
@@ -97,6 +99,12 @@ class AMDGPUMCExpr : public MCTargetExpr {
return create(VariantKind::AGVK_AlignTo, {Value, Align}, Ctx);
}
+ /// Create an expression for instruction prefetch size computation:
+ /// min(divideCeil(CodeSizeBytes, 128), (1 << FieldWidth) - 1)
+ static const AMDGPUMCExpr *createInstPrefSize(const MCExpr *CodeSizeBytes,
+ unsigned FieldWidth,
+ MCContext &Ctx);
+
static const AMDGPUMCExpr *createLit(LitModifier Lit, int64_t Value,
MCContext &Ctx);
diff --git a/llvm/lib/Target/AMDGPU/SIProgramInfo.cpp b/llvm/lib/Target/AMDGPU/SIProgramInfo.cpp
index a3f261b87e80b..abc2e01ca1df0 100644
--- a/llvm/lib/Target/AMDGPU/SIProgramInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIProgramInfo.cpp
@@ -203,9 +203,8 @@ const MCExpr *SIProgramInfo::getPGMRSrc2(CallingConv::ID CC,
return MCConstantExpr::create(0, Ctx);
}
-uint64_t SIProgramInfo::getFunctionCodeSize(const MachineFunction &MF,
- bool IsLowerBound) {
- if (!IsLowerBound && CodeSizeInBytes.has_value())
+uint64_t SIProgramInfo::getFunctionCodeSize(const MachineFunction &MF) {
+ if (CodeSizeInBytes.has_value())
return *CodeSizeInBytes;
const GCNSubtarget &STM = MF.getSubtarget<GCNSubtarget>();
@@ -214,12 +213,7 @@ uint64_t SIProgramInfo::getFunctionCodeSize(const MachineFunction &MF,
uint64_t CodeSize = 0;
for (const MachineBasicBlock &MBB : MF) {
- // The amount of padding to align code can be both underestimated and
- // overestimated. In case of inline asm used getInstSizeInBytes() will
- // return a maximum size of a single instruction, where the real size may
- // differ. At this point CodeSize may be already off.
- if (!IsLowerBound)
- CodeSize = alignTo(CodeSize, MBB.getAlignment());
+ CodeSize = alignTo(CodeSize, MBB.getAlignment());
for (const MachineInstr &MI : MBB) {
// TODO: CodeSize should account for multiple functions.
@@ -227,11 +221,6 @@ uint64_t SIProgramInfo::getFunctionCodeSize(const MachineFunction &MF,
if (MI.isMetaInstruction())
continue;
- // We cannot properly estimate inline asm size. It can be as small as zero
- // if that is just a comment.
- if (IsLowerBound && MI.isInlineAsm())
- continue;
-
CodeSize += TII->getInstSizeInBytes(MI);
}
}
diff --git a/llvm/lib/Target/AMDGPU/SIProgramInfo.h b/llvm/lib/Target/AMDGPU/SIProgramInfo.h
index 171c4a313a53b..bfd6b669531ea 100644
--- a/llvm/lib/Target/AMDGPU/SIProgramInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIProgramInfo.h
@@ -105,10 +105,7 @@ struct LLVM_EXTERNAL_VISIBILITY SIProgramInfo {
void reset(const MachineFunction &MF);
// Get function code size and cache the value.
- // If \p IsLowerBound is set it returns a minimal code size which is safe
- // to address.
- uint64_t getFunctionCodeSize(const MachineFunction &MF,
- bool IsLowerBound = false);
+ uint64_t getFunctionCodeSize(const MachineFunction &MF);
/// Compute the value of the ComputePGMRsrc1 register.
const MCExpr *getComputePGMRSrc1(const GCNSubtarget &ST,
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
index 580167076e1f0..14b9d1444a271 100644
--- a/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
@@ -1,10 +1,20 @@
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 --amdgpu-memcpy-loop-unroll=100000 < %s | FileCheck --check-prefixes=GCN,GFX11 %s
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 --amdgpu-memcpy-loop-unroll=100000 < %s | FileCheck --check-prefixes=GCN,GFX12 %s
+;; Verify that inst_pref_size resolves to the correct value in the object file.
+;; COMPUTE_PGM_RSRC3 is at offset 0x2C in each 64-byte kernel descriptor.
+;; GFX11 inst_pref_size is bits [9:4], so value N is encoded as N << 4.
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 --amdgpu-memcpy-loop-unroll=100000 -filetype=obj < %s -o %t.o
+; RUN: llvm-objdump -s -j .rodata %t.o | FileCheck --check-prefix=OBJ %s
+
+; The inst_pref_size is computed via MCExpr label subtraction
+; (code_end - func_sym), which resolves at assembly/link time.
+; In text output it appears as a symbolic expression.
+
; GCN-LABEL: .amdhsa_kernel large
-; GFX11: .amdhsa_inst_pref_size 3
+; GFX11: .amdhsa_inst_pref_size {{.*}}instprefsize({{.*}}large, 6){{.*}}
; GFX11: codeLenInByte = 3{{[0-9][0-9]$}}
-; GFX12: .amdhsa_inst_pref_size 4
+; GFX12: .amdhsa_inst_pref_size {{.*}}instprefsize({{.*}}large, 8){{.*}}
; GFX12: codeLenInByte = 4{{[0-9][0-9]$}}
define amdgpu_kernel void @large(ptr addrspace(1) %out, ptr addrspace(1) %in) {
bb:
@@ -13,20 +23,32 @@ bb:
}
; GCN-LABEL: .amdhsa_kernel small
-; GCN: .amdhsa_inst_pref_size 1
-; GCN: codeLenInByte = {{[0-9]$}}
+; GCN: .amdhsa_inst_pref_size {{.*}}instprefsize({{.*}}small, {{[0-9]+}}){{.*}}
+; GCN: codeLenInByte = {{[0-9]+$}}
define amdgpu_kernel void @small() {
bb:
ret void
}
-; Ignore inline asm in size calculation
+; Inline asm is accounted for via MCExpr label subtraction (exact code size).
+; The MCExpr resolves to the correct inst_pref_size at assembly time.
; GCN-LABEL: .amdhsa_kernel inline_asm
-; GCN: .amdhsa_inst_pref_size 1
-; GCN: codeLenInByte = {{[0-9]$}}
+; GCN: .amdhsa_inst_pref_size {{.*}}instprefsize({{.*}}inline_asm, {{[0-9]+}}){{.*}}
+; GCN: codeLenInByte = {{[0-9]+$}}
define amdgpu_kernel void @inline_asm() {
bb:
call void asm sideeffect ".fill 256, 4, 0", ""()
ret void
}
+
+;; Object file checks: verify COMPUTE_PGM_RSRC3 at offset 0x2C in each KD.
+;; COMPUTE_PGM_RSRC3 is the last dword on the 0x0020/0x0060/0x00a0 lines.
+;; GFX11 inst_pref_size is bits [9:4], so value N is encoded as N << 4.
+;;
+;; large: 348 bytes -> pref_size=3 -> 3<<4=0x30
+; OBJ: 0020 {{.*}}30000000
+;; small: 4 bytes -> pref_size=1 -> 1<<4=0x10
+; OBJ: 0060 {{.*}}10000000
+;; inline_asm: 1028 bytes -> pref_size=9 -> 9<<4=0x90
+; OBJ: 00a0 {{.*}}90000000
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
new file mode 100644
index 0000000000000..ee344bd320aaf
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
@@ -0,0 +1,83 @@
+;; Verify that inline assembly is correctly accounted for in the
+;; inst_pref_size calculation. The inst_pref_size is computed via MCExpr
+;; label subtraction (code_end - func_sym), giving exact code size.
+;; This resolves at assembly time, so we verify via object file checks.
+
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 -filetype=obj < %s -o %t.o
+; RUN: llvm-objdump -s -j .rodata %t.o | FileCheck --check-prefix=OBJ %s
+; RUN: llvm-readobj --symbols %t.o | FileCheck --check-prefix=SYM %s
+
+;; --- .fill directive: .fill 256, 4, 0 => 1024 bytes + 4 (s_endpgm) = 1028 ---
+;; pref_size = divideCeil(1028, 128) = 9
+
+; SYM: Name: test_fill
+; SYM-NEXT: Value:
+; SYM-NEXT: Size: 1028
+
+define amdgpu_kernel void @test_fill() {
+ call void asm sideeffect ".fill 256, 4, 0", ""()
+ ret void
+}
+
+;; --- .space directive: .space 1024 => 1024 bytes + 4 = 1028 ---
+;; pref_size = 9
+
+; SYM: Name: test_space
+; SYM-NEXT: Value:
+; SYM-NEXT: Size: 1028
+
+define amdgpu_kernel void @test_space() {
+ call void asm sideeffect ".space 1024", ""()
+ ret void
+}
+
+;; --- Instructions: 32 x s_nop (4 bytes each) = 128 + 4 = 132 ---
+;; pref_size = divideCeil(132, 128) = 2
+
+; SYM: Name: test_instructions
+; SYM-NEXT: Value:
+; SYM-NEXT: Size: 132
+
+define amdgpu_kernel void @test_instructions() {
+ call void asm sideeffect "s_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0", ""()
+ ret void
+}
+
+;; --- Comments emit no bytes: only s_endpgm = 4 bytes ---
+;; pref_size = 1
+
+; SYM: Name: test_comments
+; SYM-NEXT: Value:
+; SYM-NEXT: Size: 4
+
+define amdgpu_kernel void @test_comments() {
+ call void asm sideeffect "; comment 1\0A; comment 2\0A; comment 3", ""()
+ ret void
+}
+
+;; --- Empty inline asm: only s_endpgm = 4 bytes ---
+;; pref_size = 1
+
+; SYM: Name: test_empty_asm
+; SYM-NEXT: Value:
+; SYM-NEXT: Size: 4
+
+define amdgpu_kernel void @test_empty_asm() {
+ call void asm sideeffect "", ""()
+ ret void
+}
+
+;; Object file checks: verify COMPUTE_PGM_RSRC3 at offset 0x2C in each
+;; 64-byte kernel descriptor. GFX11 inst_pref_size is bits [9:4].
+;;
+;; test_fill: 1028 bytes -> pref_size=9 -> 9<<4 = 0x90
+;; test_space: 1028 bytes -> pref_size=9 -> 9<<4 = 0x90
+;; test_instructions: 132 bytes -> pref_size=2 -> 2<<4 = 0x20
+;; test_comments: 4 bytes -> pref_size=1 -> 1<<4 = 0x10
+;; test_empty_asm: 4 bytes -> pref_size=1 -> 1<<4 = 0x10
+
+; OBJ: 0020 {{.*}}90000000
+; OBJ: 0060 {{.*}}90000000
+; OBJ: 00a0 {{.*}}20000000
+; OBJ: 00e0 {{.*}}10000000
+; OBJ: 0120 {{.*}}10000000
>From d3d85316a9d661659695404c61e5abf0bc412339 Mon Sep 17 00:00:00 2001
From: Adel Ejjeh <adel.ejjeh at amd.com>
Date: Wed, 15 Apr 2026 15:24:40 -0500
Subject: [PATCH 02/11] Address review feedback: use exact MCExpr CHECK lines,
add GFX12 obj checks, add more detailed comments
Co-Authored-By: Claude Opus 4 (1M context) <noreply at anthropic.com>
---
.../test/CodeGen/AMDGPU/inst-prefetch-hint.ll | 58 +++++++++++--------
.../AMDGPU/inst-prefetch-inline-asm.ll | 40 +++++++------
2 files changed, 56 insertions(+), 42 deletions(-)
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
index 14b9d1444a271..2127c69a6f2fd 100644
--- a/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
@@ -3,19 +3,29 @@
;; Verify that inst_pref_size resolves to the correct value in the object file.
;; COMPUTE_PGM_RSRC3 is at offset 0x2C in each 64-byte kernel descriptor.
-;; GFX11 inst_pref_size is bits [9:4], so value N is encoded as N << 4.
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 --amdgpu-memcpy-loop-unroll=100000 -filetype=obj < %s -o %t.o
-; RUN: llvm-objdump -s -j .rodata %t.o | FileCheck --check-prefix=OBJ %s
+;; inst_pref_size is bits [9:4] on GFX11 (6-bit) and bits [11:4] on GFX12+ (8-bit).
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 --amdgpu-memcpy-loop-unroll=100000 -filetype=obj < %s -o %t.gfx11.o
+; RUN: llvm-objdump -s -j .rodata %t.gfx11.o | FileCheck --check-prefix=OBJ-GFX11 %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 --amdgpu-memcpy-loop-unroll=100000 -filetype=obj < %s -o %t.gfx12.o
+; RUN: llvm-objdump -s -j .rodata %t.gfx12.o | FileCheck --check-prefix=OBJ-GFX12 %s
-; The inst_pref_size is computed via MCExpr label subtraction
-; (code_end - func_sym), which resolves at assembly/link time.
-; In text output it appears as a symbolic expression.
+; The inst_pref_size is computed via MCExpr label subtraction, resolved at
+; assembly/link time. In text output it appears as a symbolic expression:
+; ((instprefsize(<code_size>, <field_width>) << 4) & <mask>) >> 4
+; where:
+; <code_size> = .Lfunc_endN - func_sym (exact function code size in bytes)
+; <field_width> = bit width of the inst_pref_size field (6 for GFX11, 8 for GFX12+)
+; instprefsize = min(divideCeil(code_size, 128), (1 << field_width) - 1)
+; << 4, & mask, >> 4 = bit-field insertion/extraction within COMPUTE_PGM_RSRC3
; GCN-LABEL: .amdhsa_kernel large
-; GFX11: .amdhsa_inst_pref_size {{.*}}instprefsize({{.*}}large, 6){{.*}}
-; GFX11: codeLenInByte = 3{{[0-9][0-9]$}}
-; GFX12: .amdhsa_inst_pref_size {{.*}}instprefsize({{.*}}large, 8){{.*}}
-; GFX12: codeLenInByte = 4{{[0-9][0-9]$}}
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-large, 6)<<4)&1008)>>4
+; GFX11: codeLenInByte = 348
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-large, 8)<<4)&4080)>>4
+; GFX12: codeLenInByte = 476
+;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C: gfx11 pref=3 (0x30), gfx12 pref=4 (0x40)
+; OBJ-GFX11: 0020 {{.*}}30000000
+; OBJ-GFX12: 0020 {{.*}}40000000
define amdgpu_kernel void @large(ptr addrspace(1) %out, ptr addrspace(1) %in) {
bb:
call void @llvm.memcpy.p1.p3.i32(ptr addrspace(1) %out, ptr addrspace(1) %in, i32 256, i1 false)
@@ -23,8 +33,12 @@ bb:
}
; GCN-LABEL: .amdhsa_kernel small
-; GCN: .amdhsa_inst_pref_size {{.*}}instprefsize({{.*}}small, {{[0-9]+}}){{.*}}
-; GCN: codeLenInByte = {{[0-9]+$}}
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-small, 6)<<4)&1008)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-small, 8)<<4)&4080)>>4
+; GCN: codeLenInByte = 4
+;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C: pref=1 (0x10) for both
+; OBJ-GFX11: 0060 {{.*}}10000000
+; OBJ-GFX12: 0060 {{.*}}10000000
define amdgpu_kernel void @small() {
bb:
ret void
@@ -34,21 +48,15 @@ bb:
; The MCExpr resolves to the correct inst_pref_size at assembly time.
; GCN-LABEL: .amdhsa_kernel inline_asm
-; GCN: .amdhsa_inst_pref_size {{.*}}instprefsize({{.*}}inline_asm, {{[0-9]+}}){{.*}}
-; GCN: codeLenInByte = {{[0-9]+$}}
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-inline_asm, 6)<<4)&1008)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-inline_asm, 8)<<4)&4080)>>4
+; GCN: codeLenInByte = 24
+;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC: pref=9 (0x90) for both
+;; (.fill 256, 4, 0 = 1024 bytes + 4 s_endpgm = 1028 -> divideCeil(1028,128) = 9)
+; OBJ-GFX11: 00a0 {{.*}}90000000
+; OBJ-GFX12: 00a0 {{.*}}90000000
define amdgpu_kernel void @inline_asm() {
bb:
call void asm sideeffect ".fill 256, 4, 0", ""()
ret void
}
-
-;; Object file checks: verify COMPUTE_PGM_RSRC3 at offset 0x2C in each KD.
-;; COMPUTE_PGM_RSRC3 is the last dword on the 0x0020/0x0060/0x00a0 lines.
-;; GFX11 inst_pref_size is bits [9:4], so value N is encoded as N << 4.
-;;
-;; large: 348 bytes -> pref_size=3 -> 3<<4=0x30
-; OBJ: 0020 {{.*}}30000000
-;; small: 4 bytes -> pref_size=1 -> 1<<4=0x10
-; OBJ: 0060 {{.*}}10000000
-;; inline_asm: 1028 bytes -> pref_size=9 -> 9<<4=0x90
-; OBJ: 00a0 {{.*}}90000000
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
index ee344bd320aaf..ca0c928cc0ab6 100644
--- a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
@@ -1,8 +1,9 @@
;; Verify that inline assembly is correctly accounted for in the
;; inst_pref_size calculation. The inst_pref_size is computed via MCExpr
-;; label subtraction (code_end - func_sym), giving exact code size.
-;; This resolves at assembly time, so we verify via object file checks.
+;; label subtraction (.Lfunc_end - func_sym), giving exact code size.
+;; See inst-prefetch-hint.ll for explanation of the instprefsize expression.
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 < %s | FileCheck --check-prefix=ASM %s
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 -filetype=obj < %s -o %t.o
; RUN: llvm-objdump -s -j .rodata %t.o | FileCheck --check-prefix=OBJ %s
; RUN: llvm-readobj --symbols %t.o | FileCheck --check-prefix=SYM %s
@@ -10,9 +11,13 @@
;; --- .fill directive: .fill 256, 4, 0 => 1024 bytes + 4 (s_endpgm) = 1028 ---
;; pref_size = divideCeil(1028, 128) = 9
+; ASM-LABEL: .amdhsa_kernel test_fill
+; ASM: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-test_fill, 6)<<4)&1008)>>4
; SYM: Name: test_fill
; SYM-NEXT: Value:
; SYM-NEXT: Size: 1028
+;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C: pref_size=9 -> 9<<4 = 0x90
+; OBJ: 0020 {{.*}}90000000
define amdgpu_kernel void @test_fill() {
call void asm sideeffect ".fill 256, 4, 0", ""()
@@ -22,9 +27,13 @@ define amdgpu_kernel void @test_fill() {
;; --- .space directive: .space 1024 => 1024 bytes + 4 = 1028 ---
;; pref_size = 9
+; ASM-LABEL: .amdhsa_kernel test_space
+; ASM: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-test_space, 6)<<4)&1008)>>4
; SYM: Name: test_space
; SYM-NEXT: Value:
; SYM-NEXT: Size: 1028
+;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C: pref_size=9 -> 9<<4 = 0x90
+; OBJ: 0060 {{.*}}90000000
define amdgpu_kernel void @test_space() {
call void asm sideeffect ".space 1024", ""()
@@ -34,9 +43,13 @@ define amdgpu_kernel void @test_space() {
;; --- Instructions: 32 x s_nop (4 bytes each) = 128 + 4 = 132 ---
;; pref_size = divideCeil(132, 128) = 2
+; ASM-LABEL: .amdhsa_kernel test_instructions
+; ASM: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-test_instructions, 6)<<4)&1008)>>4
; SYM: Name: test_instructions
; SYM-NEXT: Value:
; SYM-NEXT: Size: 132
+;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC: pref_size=2 -> 2<<4 = 0x20
+; OBJ: 00a0 {{.*}}20000000
define amdgpu_kernel void @test_instructions() {
call void asm sideeffect "s_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0", ""()
@@ -46,9 +59,13 @@ define amdgpu_kernel void @test_instructions() {
;; --- Comments emit no bytes: only s_endpgm = 4 bytes ---
;; pref_size = 1
+; ASM-LABEL: .amdhsa_kernel test_comments
+; ASM: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end3-test_comments, 6)<<4)&1008)>>4
; SYM: Name: test_comments
; SYM-NEXT: Value:
; SYM-NEXT: Size: 4
+;; Object: kernel descriptor at 0xC0, COMPUTE_PGM_RSRC3 at 0xEC: pref_size=1 -> 1<<4 = 0x10
+; OBJ: 00e0 {{.*}}10000000
define amdgpu_kernel void @test_comments() {
call void asm sideeffect "; comment 1\0A; comment 2\0A; comment 3", ""()
@@ -58,26 +75,15 @@ define amdgpu_kernel void @test_comments() {
;; --- Empty inline asm: only s_endpgm = 4 bytes ---
;; pref_size = 1
+; ASM-LABEL: .amdhsa_kernel test_empty_asm
+; ASM: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end4-test_empty_asm, 6)<<4)&1008)>>4
; SYM: Name: test_empty_asm
; SYM-NEXT: Value:
; SYM-NEXT: Size: 4
+;; Object: kernel descriptor at 0x100, COMPUTE_PGM_RSRC3 at 0x12C: pref_size=1 -> 1<<4 = 0x10
+; OBJ: 0120 {{.*}}10000000
define amdgpu_kernel void @test_empty_asm() {
call void asm sideeffect "", ""()
ret void
}
-
-;; Object file checks: verify COMPUTE_PGM_RSRC3 at offset 0x2C in each
-;; 64-byte kernel descriptor. GFX11 inst_pref_size is bits [9:4].
-;;
-;; test_fill: 1028 bytes -> pref_size=9 -> 9<<4 = 0x90
-;; test_space: 1028 bytes -> pref_size=9 -> 9<<4 = 0x90
-;; test_instructions: 132 bytes -> pref_size=2 -> 2<<4 = 0x20
-;; test_comments: 4 bytes -> pref_size=1 -> 1<<4 = 0x10
-;; test_empty_asm: 4 bytes -> pref_size=1 -> 1<<4 = 0x10
-
-; OBJ: 0020 {{.*}}90000000
-; OBJ: 0060 {{.*}}90000000
-; OBJ: 00a0 {{.*}}20000000
-; OBJ: 00e0 {{.*}}10000000
-; OBJ: 0120 {{.*}}10000000
>From 72cc874752104a33be2ec95b5bc05de3d36dd6ff Mon Sep 17 00:00:00 2001
From: Adel Ejjeh <adel.ejjeh at amd.com>
Date: Wed, 15 Apr 2026 16:19:27 -0500
Subject: [PATCH 03/11] Keep inst_pref_size separate from ComputePGMRSrc3 to
fix s-barrier-lowering regression
Co-Authored-By: Claude Opus 4 (1M context) <noreply at anthropic.com>
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 33 +++++++++++-------
.../MCTargetDesc/AMDGPUMCKernelDescriptor.h | 8 +++++
.../MCTargetDesc/AMDGPUTargetStreamer.cpp | 29 ++++++++++------
.../test/CodeGen/AMDGPU/inst-prefetch-hint.ll | 34 ++++++++++---------
.../AMDGPU/inst-prefetch-inline-asm.ll | 25 ++++++++------
5 files changed, 80 insertions(+), 49 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index c57e592afc2ad..a6f13b9d35cb2 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -255,29 +255,37 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
const GCNSubtarget &STM = MF->getSubtarget<GCNSubtarget>();
MCContext &Ctx = MF->getContext();
+ AMDGPU::MCKernelDescriptor KD =
+ getAmdhsaKernelDescriptor(*MF, CurrentProgramInfo);
+
// Compute inst_pref_size using MCExpr label subtraction for exact code
- // size. (Lfunc_end - func_sym) gives the exact function code size in bytes.
+ // size. At this point .Lfunc_end has been emitted (by the base AsmPrinter)
+ // right after the function code, so (Lfunc_end - func_sym) gives the
+ // exact function code size in bytes.
+ // We store it as a separate KD field rather than OR'ing into
+ // compute_pgm_rsrc3, because the label subtraction MCExpr is unresolvable
+ // in text mode and would prevent printing of other fields (e.g.
+ // named_barrier_count) that share the same register.
if (isGFX11Plus(STM)) {
const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
- uint32_t Field, Shift, Width;
+ uint32_t Width;
if (isGFX11(STM)) {
- Field = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE;
- Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT;
+ KD.inst_pref_size_mask = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE;
+ KD.inst_pref_size_shift =
+ amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT;
Width = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_WIDTH;
} else {
- Field = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE;
- Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT;
+ KD.inst_pref_size_mask =
+ amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE;
+ KD.inst_pref_size_shift =
+ amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT;
Width = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_WIDTH;
}
- const MCExpr *InstPrefSizeExpr =
+ KD.inst_pref_size =
AMDGPUMCExpr::createInstPrefSize(CodeSizeExpr, Width, Ctx);
-
- CurrentProgramInfo.ComputePGMRSrc3 =
- setBits(CurrentProgramInfo.ComputePGMRSrc3, InstPrefSizeExpr, Field,
- Shift, Ctx);
}
auto &Streamer = getTargetStreamer()->getStreamer();
@@ -296,8 +304,7 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
SmallString<128> KernelName;
getNameWithPrefix(KernelName, &MF->getFunction());
getTargetStreamer()->EmitAmdhsaKernelDescriptor(
- STM, KernelName, getAmdhsaKernelDescriptor(*MF, CurrentProgramInfo),
- CurrentProgramInfo.NumVGPRsForWavesPerEU,
+ STM, KernelName, KD, CurrentProgramInfo.NumVGPRsForWavesPerEU,
MCBinaryExpr::createSub(
CurrentProgramInfo.NumSGPRsForWavesPerEU,
AMDGPUMCExpr::createExtraSGPRs(
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCKernelDescriptor.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCKernelDescriptor.h
index 26958ac8b9ee1..d8bd41cdff5ce 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCKernelDescriptor.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCKernelDescriptor.h
@@ -34,6 +34,14 @@ struct MCKernelDescriptor {
const MCExpr *kernel_code_properties = nullptr;
const MCExpr *kernarg_preload = nullptr;
+ /// Instruction prefetch size, kept separate from compute_pgm_rsrc3 to avoid
+ /// contaminating the register MCExpr with an unresolvable label subtraction
+ /// (which would prevent text-mode printing of other fields in the register).
+ /// The ELF streamer OR's this into compute_pgm_rsrc3 when emitting bytes.
+ const MCExpr *inst_pref_size = nullptr;
+ uint32_t inst_pref_size_shift = 0;
+ uint32_t inst_pref_size_mask = 0;
+
static MCKernelDescriptor
getDefaultAmdhsaKernelDescriptor(const MCSubtargetInfo *STI, MCContext &Ctx);
// MCExpr for:
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
index 47733494d421b..ad1a7e533b60d 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
@@ -607,17 +607,17 @@ void AMDGPUTargetAsmStreamer::EmitAmdhsaKernelDescriptor(
amdhsa::COMPUTE_PGM_RSRC3_GFX10_GFX11_SHARED_VGPR_COUNT,
".amdhsa_shared_vgpr_count");
}
- if (IVersion.Major == 11) {
- PrintField(KD.compute_pgm_rsrc3,
- amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT,
- amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE,
- ".amdhsa_inst_pref_size");
+ if (IVersion.Major >= 11) {
+ OS << "\t\t.amdhsa_inst_pref_size ";
+ if (KD.inst_pref_size) {
+ const MCExpr *New = foldAMDGPUMCExpr(KD.inst_pref_size, getContext());
+ printAMDGPUMCExpr(New, OS, MAI);
+ } else {
+ OS << 0;
+ }
+ OS << '\n';
}
if (IVersion.Major >= 12) {
- PrintField(KD.compute_pgm_rsrc3,
- amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT,
- amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE,
- ".amdhsa_inst_pref_size");
PrintField(KD.compute_pgm_rsrc1,
amdhsa::COMPUTE_PGM_RSRC1_GFX12_PLUS_ENABLE_WG_RR_EN_SHIFT,
amdhsa::COMPUTE_PGM_RSRC1_GFX12_PLUS_ENABLE_WG_RR_EN,
@@ -1051,7 +1051,16 @@ void AMDGPUTargetELFStreamer::EmitAmdhsaKernelDescriptor(
sizeof(amdhsa::kernel_descriptor_t::kernel_code_entry_byte_offset));
for (uint32_t i = 0; i < sizeof(amdhsa::kernel_descriptor_t::reserved1); ++i)
Streamer.emitInt8(0u);
- Streamer.emitValue(KernelDescriptor.compute_pgm_rsrc3,
+ // OR inst_pref_size into compute_pgm_rsrc3 for the binary encoding.
+ // This is kept separate in the KD struct to avoid making the MCExpr
+ // unresolvable in text mode (see AMDGPUAsmPrinter::endFunction).
+ const MCExpr *Rsrc3 = KernelDescriptor.compute_pgm_rsrc3;
+ if (KernelDescriptor.inst_pref_size) {
+ MCKernelDescriptor::bits_set(Rsrc3, KernelDescriptor.inst_pref_size,
+ KernelDescriptor.inst_pref_size_shift,
+ KernelDescriptor.inst_pref_size_mask, Context);
+ }
+ Streamer.emitValue(Rsrc3,
sizeof(amdhsa::kernel_descriptor_t::compute_pgm_rsrc3));
Streamer.emitValue(KernelDescriptor.compute_pgm_rsrc1,
sizeof(amdhsa::kernel_descriptor_t::compute_pgm_rsrc1));
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
index 2127c69a6f2fd..99adbce0d37dc 100644
--- a/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
@@ -10,20 +10,20 @@
; RUN: llvm-objdump -s -j .rodata %t.gfx12.o | FileCheck --check-prefix=OBJ-GFX12 %s
; The inst_pref_size is computed via MCExpr label subtraction, resolved at
-; assembly/link time. In text output it appears as a symbolic expression:
-; ((instprefsize(<code_size>, <field_width>) << 4) & <mask>) >> 4
+; assembly/link time. In text output it appears as:
+; instprefsize(<code_size>, <field_width>)
; where:
; <code_size> = .Lfunc_endN - func_sym (exact function code size in bytes)
; <field_width> = bit width of the inst_pref_size field (6 for GFX11, 8 for GFX12+)
; instprefsize = min(divideCeil(code_size, 128), (1 << field_width) - 1)
-; << 4, & mask, >> 4 = bit-field insertion/extraction within COMPUTE_PGM_RSRC3
; GCN-LABEL: .amdhsa_kernel large
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-large, 6)<<4)&1008)>>4
-; GFX11: codeLenInByte = 348
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-large, 8)<<4)&4080)>>4
-; GFX12: codeLenInByte = 476
-;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C: gfx11 pref=3 (0x30), gfx12 pref=4 (0x40)
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-large, 6)
+; GFX11: codeLenInByte = {{[0-9]+}}
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-large, 8)
+; GFX12: codeLenInByte = {{[0-9]+}}
+;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C:
+;; gfx11 pref=3 (0x30), gfx12 pref=4 (0x40)
; OBJ-GFX11: 0020 {{.*}}30000000
; OBJ-GFX12: 0020 {{.*}}40000000
define amdgpu_kernel void @large(ptr addrspace(1) %out, ptr addrspace(1) %in) {
@@ -33,10 +33,11 @@ bb:
}
; GCN-LABEL: .amdhsa_kernel small
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-small, 6)<<4)&1008)>>4
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-small, 8)<<4)&4080)>>4
-; GCN: codeLenInByte = 4
-;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C: pref=1 (0x10) for both
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-small, 6)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-small, 8)
+; GCN: codeLenInByte = {{[0-9]+}}
+;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C:
+;; pref=1 (0x10) for both
; OBJ-GFX11: 0060 {{.*}}10000000
; OBJ-GFX12: 0060 {{.*}}10000000
define amdgpu_kernel void @small() {
@@ -48,10 +49,11 @@ bb:
; The MCExpr resolves to the correct inst_pref_size at assembly time.
; GCN-LABEL: .amdhsa_kernel inline_asm
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-inline_asm, 6)<<4)&1008)>>4
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-inline_asm, 8)<<4)&4080)>>4
-; GCN: codeLenInByte = 24
-;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC: pref=9 (0x90) for both
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-inline_asm, 6)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-inline_asm, 8)
+; GCN: codeLenInByte = {{[0-9]+}}
+;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC:
+;; pref=9 (0x90) for both
;; (.fill 256, 4, 0 = 1024 bytes + 4 s_endpgm = 1028 -> divideCeil(1028,128) = 9)
; OBJ-GFX11: 00a0 {{.*}}90000000
; OBJ-GFX12: 00a0 {{.*}}90000000
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
index ca0c928cc0ab6..2e3ad07c89f3d 100644
--- a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
@@ -12,11 +12,12 @@
;; pref_size = divideCeil(1028, 128) = 9
; ASM-LABEL: .amdhsa_kernel test_fill
-; ASM: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-test_fill, 6)<<4)&1008)>>4
+; ASM: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-test_fill, 6)
; SYM: Name: test_fill
; SYM-NEXT: Value:
; SYM-NEXT: Size: 1028
-;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C: pref_size=9 -> 9<<4 = 0x90
+;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C:
+;; pref_size=9 -> 9<<4 = 0x90
; OBJ: 0020 {{.*}}90000000
define amdgpu_kernel void @test_fill() {
@@ -28,11 +29,12 @@ define amdgpu_kernel void @test_fill() {
;; pref_size = 9
; ASM-LABEL: .amdhsa_kernel test_space
-; ASM: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-test_space, 6)<<4)&1008)>>4
+; ASM: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-test_space, 6)
; SYM: Name: test_space
; SYM-NEXT: Value:
; SYM-NEXT: Size: 1028
-;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C: pref_size=9 -> 9<<4 = 0x90
+;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C:
+;; pref_size=9 -> 9<<4 = 0x90
; OBJ: 0060 {{.*}}90000000
define amdgpu_kernel void @test_space() {
@@ -44,11 +46,12 @@ define amdgpu_kernel void @test_space() {
;; pref_size = divideCeil(132, 128) = 2
; ASM-LABEL: .amdhsa_kernel test_instructions
-; ASM: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-test_instructions, 6)<<4)&1008)>>4
+; ASM: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-test_instructions, 6)
; SYM: Name: test_instructions
; SYM-NEXT: Value:
; SYM-NEXT: Size: 132
-;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC: pref_size=2 -> 2<<4 = 0x20
+;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC:
+;; pref_size=2 -> 2<<4 = 0x20
; OBJ: 00a0 {{.*}}20000000
define amdgpu_kernel void @test_instructions() {
@@ -60,11 +63,12 @@ define amdgpu_kernel void @test_instructions() {
;; pref_size = 1
; ASM-LABEL: .amdhsa_kernel test_comments
-; ASM: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end3-test_comments, 6)<<4)&1008)>>4
+; ASM: .amdhsa_inst_pref_size instprefsize(.Lfunc_end3-test_comments, 6)
; SYM: Name: test_comments
; SYM-NEXT: Value:
; SYM-NEXT: Size: 4
-;; Object: kernel descriptor at 0xC0, COMPUTE_PGM_RSRC3 at 0xEC: pref_size=1 -> 1<<4 = 0x10
+;; Object: kernel descriptor at 0xC0, COMPUTE_PGM_RSRC3 at 0xEC:
+;; pref_size=1 -> 1<<4 = 0x10
; OBJ: 00e0 {{.*}}10000000
define amdgpu_kernel void @test_comments() {
@@ -76,11 +80,12 @@ define amdgpu_kernel void @test_comments() {
;; pref_size = 1
; ASM-LABEL: .amdhsa_kernel test_empty_asm
-; ASM: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end4-test_empty_asm, 6)<<4)&1008)>>4
+; ASM: .amdhsa_inst_pref_size instprefsize(.Lfunc_end4-test_empty_asm, 6)
; SYM: Name: test_empty_asm
; SYM-NEXT: Value:
; SYM-NEXT: Size: 4
-;; Object: kernel descriptor at 0x100, COMPUTE_PGM_RSRC3 at 0x12C: pref_size=1 -> 1<<4 = 0x10
+;; Object: kernel descriptor at 0x100, COMPUTE_PGM_RSRC3 at 0x12C:
+;; pref_size=1 -> 1<<4 = 0x10
; OBJ: 0120 {{.*}}10000000
define amdgpu_kernel void @test_empty_asm() {
>From c2773e44b2319f1c321269a2cc4fb39f43f02d46 Mon Sep 17 00:00:00 2001
From: Adel Ejjeh <adel.ejjeh at amd.com>
Date: Wed, 15 Apr 2026 16:45:25 -0500
Subject: [PATCH 04/11] Fix MC assembler round-trip: fall back to PrintField
when inst_pref_size is not set
Co-Authored-By: Claude Opus 4 (1M context) <noreply at anthropic.com>
---
.../MCTargetDesc/AMDGPUTargetStreamer.cpp | 24 +++-
.../AMDGPU/inst-prefetch-inline-asm.ll | 128 +++++++++++++-----
2 files changed, 114 insertions(+), 38 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
index ad1a7e533b60d..d75f99bd5452c 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
@@ -607,17 +607,33 @@ void AMDGPUTargetAsmStreamer::EmitAmdhsaKernelDescriptor(
amdhsa::COMPUTE_PGM_RSRC3_GFX10_GFX11_SHARED_VGPR_COUNT,
".amdhsa_shared_vgpr_count");
}
- if (IVersion.Major >= 11) {
- OS << "\t\t.amdhsa_inst_pref_size ";
+ if (IVersion.Major == 11) {
if (KD.inst_pref_size) {
+ // CodeGen path: print the MCExpr directly (label subtraction).
+ OS << "\t\t.amdhsa_inst_pref_size ";
const MCExpr *New = foldAMDGPUMCExpr(KD.inst_pref_size, getContext());
printAMDGPUMCExpr(New, OS, MAI);
+ OS << '\n';
} else {
- OS << 0;
+ // MC assembler path: extract from compute_pgm_rsrc3.
+ PrintField(KD.compute_pgm_rsrc3,
+ amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT,
+ amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE,
+ ".amdhsa_inst_pref_size");
}
- OS << '\n';
}
if (IVersion.Major >= 12) {
+ if (KD.inst_pref_size) {
+ OS << "\t\t.amdhsa_inst_pref_size ";
+ const MCExpr *New = foldAMDGPUMCExpr(KD.inst_pref_size, getContext());
+ printAMDGPUMCExpr(New, OS, MAI);
+ OS << '\n';
+ } else {
+ PrintField(KD.compute_pgm_rsrc3,
+ amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT,
+ amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE,
+ ".amdhsa_inst_pref_size");
+ }
PrintField(KD.compute_pgm_rsrc1,
amdhsa::COMPUTE_PGM_RSRC1_GFX12_PLUS_ENABLE_WG_RR_EN_SHIFT,
amdhsa::COMPUTE_PGM_RSRC1_GFX12_PLUS_ENABLE_WG_RR_EN,
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
index 2e3ad07c89f3d..1dff24e6c234f 100644
--- a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
@@ -3,22 +3,24 @@
;; label subtraction (.Lfunc_end - func_sym), giving exact code size.
;; See inst-prefetch-hint.ll for explanation of the instprefsize expression.
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 < %s | FileCheck --check-prefix=ASM %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 -filetype=obj < %s -o %t.o
-; RUN: llvm-objdump -s -j .rodata %t.o | FileCheck --check-prefix=OBJ %s
-; RUN: llvm-readobj --symbols %t.o | FileCheck --check-prefix=SYM %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 < %s | FileCheck --check-prefix=GFX11 %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 < %s | FileCheck --check-prefix=GFX12 %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 -filetype=obj < %s -o %t.gfx11.o
+; RUN: llvm-objdump -s -j .rodata %t.gfx11.o | FileCheck --check-prefix=OBJ-GFX11 %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -filetype=obj < %s -o %t.gfx12.o
+; RUN: llvm-objdump -s -j .rodata %t.gfx12.o | FileCheck --check-prefix=OBJ-GFX12 %s
;; --- .fill directive: .fill 256, 4, 0 => 1024 bytes + 4 (s_endpgm) = 1028 ---
;; pref_size = divideCeil(1028, 128) = 9
-; ASM-LABEL: .amdhsa_kernel test_fill
-; ASM: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-test_fill, 6)
-; SYM: Name: test_fill
-; SYM-NEXT: Value:
-; SYM-NEXT: Size: 1028
+; GFX11-LABEL: .amdhsa_kernel test_fill
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-test_fill, 6)
+; GFX12-LABEL: .amdhsa_kernel test_fill
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-test_fill, 8)
;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C:
;; pref_size=9 -> 9<<4 = 0x90
-; OBJ: 0020 {{.*}}90000000
+; OBJ-GFX11: 0020 {{.*}}90000000
+; OBJ-GFX12: 0020 {{.*}}90000000
define amdgpu_kernel void @test_fill() {
call void asm sideeffect ".fill 256, 4, 0", ""()
@@ -28,14 +30,14 @@ define amdgpu_kernel void @test_fill() {
;; --- .space directive: .space 1024 => 1024 bytes + 4 = 1028 ---
;; pref_size = 9
-; ASM-LABEL: .amdhsa_kernel test_space
-; ASM: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-test_space, 6)
-; SYM: Name: test_space
-; SYM-NEXT: Value:
-; SYM-NEXT: Size: 1028
+; GFX11-LABEL: .amdhsa_kernel test_space
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-test_space, 6)
+; GFX12-LABEL: .amdhsa_kernel test_space
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-test_space, 8)
;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C:
;; pref_size=9 -> 9<<4 = 0x90
-; OBJ: 0060 {{.*}}90000000
+; OBJ-GFX11: 0060 {{.*}}90000000
+; OBJ-GFX12: 0060 {{.*}}90000000
define amdgpu_kernel void @test_space() {
call void asm sideeffect ".space 1024", ""()
@@ -45,14 +47,14 @@ define amdgpu_kernel void @test_space() {
;; --- Instructions: 32 x s_nop (4 bytes each) = 128 + 4 = 132 ---
;; pref_size = divideCeil(132, 128) = 2
-; ASM-LABEL: .amdhsa_kernel test_instructions
-; ASM: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-test_instructions, 6)
-; SYM: Name: test_instructions
-; SYM-NEXT: Value:
-; SYM-NEXT: Size: 132
+; GFX11-LABEL: .amdhsa_kernel test_instructions
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-test_instructions, 6)
+; GFX12-LABEL: .amdhsa_kernel test_instructions
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-test_instructions, 8)
;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC:
;; pref_size=2 -> 2<<4 = 0x20
-; OBJ: 00a0 {{.*}}20000000
+; OBJ-GFX11: 00a0 {{.*}}20000000
+; OBJ-GFX12: 00a0 {{.*}}20000000
define amdgpu_kernel void @test_instructions() {
call void asm sideeffect "s_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0\0As_nop 0", ""()
@@ -62,14 +64,14 @@ define amdgpu_kernel void @test_instructions() {
;; --- Comments emit no bytes: only s_endpgm = 4 bytes ---
;; pref_size = 1
-; ASM-LABEL: .amdhsa_kernel test_comments
-; ASM: .amdhsa_inst_pref_size instprefsize(.Lfunc_end3-test_comments, 6)
-; SYM: Name: test_comments
-; SYM-NEXT: Value:
-; SYM-NEXT: Size: 4
+; GFX11-LABEL: .amdhsa_kernel test_comments
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end3-test_comments, 6)
+; GFX12-LABEL: .amdhsa_kernel test_comments
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end3-test_comments, 8)
;; Object: kernel descriptor at 0xC0, COMPUTE_PGM_RSRC3 at 0xEC:
;; pref_size=1 -> 1<<4 = 0x10
-; OBJ: 00e0 {{.*}}10000000
+; OBJ-GFX11: 00e0 {{.*}}10000000
+; OBJ-GFX12: 00e0 {{.*}}10000000
define amdgpu_kernel void @test_comments() {
call void asm sideeffect "; comment 1\0A; comment 2\0A; comment 3", ""()
@@ -79,16 +81,74 @@ define amdgpu_kernel void @test_comments() {
;; --- Empty inline asm: only s_endpgm = 4 bytes ---
;; pref_size = 1
-; ASM-LABEL: .amdhsa_kernel test_empty_asm
-; ASM: .amdhsa_inst_pref_size instprefsize(.Lfunc_end4-test_empty_asm, 6)
-; SYM: Name: test_empty_asm
-; SYM-NEXT: Value:
-; SYM-NEXT: Size: 4
+; GFX11-LABEL: .amdhsa_kernel test_empty_asm
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end4-test_empty_asm, 6)
+; GFX12-LABEL: .amdhsa_kernel test_empty_asm
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end4-test_empty_asm, 8)
;; Object: kernel descriptor at 0x100, COMPUTE_PGM_RSRC3 at 0x12C:
;; pref_size=1 -> 1<<4 = 0x10
-; OBJ: 0120 {{.*}}10000000
+; OBJ-GFX11: 0120 {{.*}}10000000
+; OBJ-GFX12: 0120 {{.*}}10000000
define amdgpu_kernel void @test_empty_asm() {
call void asm sideeffect "", ""()
ret void
}
+
+;; --- Multiple inline asm blocks: .fill (512) + .space (512) + s_endpgm (4) = 1028 ---
+;; pref_size = divideCeil(1028, 128) = 9
+
+; GFX11-LABEL: .amdhsa_kernel test_multiple_asm
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end5-test_multiple_asm, 6)
+; GFX12-LABEL: .amdhsa_kernel test_multiple_asm
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end5-test_multiple_asm, 8)
+;; Object: kernel descriptor at 0x140, COMPUTE_PGM_RSRC3 at 0x16C:
+;; pref_size=9 -> 9<<4 = 0x90
+; OBJ-GFX11: 0160 {{.*}}90000000
+; OBJ-GFX12: 0160 {{.*}}90000000
+
+define amdgpu_kernel void @test_multiple_asm() {
+ call void asm sideeffect ".fill 128, 4, 0", ""()
+ call void asm sideeffect ".space 512", ""()
+ ret void
+}
+
+;; --- Large function that exceeds GFX11 6-bit field max (63) ---
+;; .fill 2048, 4, 0 = 8192 bytes + 4 = 8196 bytes
+;; divideCeil(8196, 128) = 65, but GFX11 max = (1<<6)-1 = 63
+;; pref_size should clamp to 63
+
+; GFX11-LABEL: .amdhsa_kernel test_clamping
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end6-test_clamping, 6)
+; GFX12-LABEL: .amdhsa_kernel test_clamping
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end6-test_clamping, 8)
+;; Object: kernel descriptor at 0x180, COMPUTE_PGM_RSRC3 at 0x1AC:
+;; gfx11: clamped to 63 -> 63<<4 = 0x3F0
+;; gfx12: no clamping, 65 -> 65<<4 = 0x410
+; OBJ-GFX11: 01a0 {{.*}}f0030000
+; OBJ-GFX12: 01a0 {{.*}}10040000
+
+define amdgpu_kernel void @test_clamping() {
+ call void asm sideeffect ".fill 2048, 4, 0", ""()
+ ret void
+}
+
+;; --- Large function that exceeds both GFX11 and GFX12 field max ---
+;; .fill 8192, 4, 0 = 32768 bytes + 4 = 32772 bytes
+;; divideCeil(32772, 128) = 257
+;; GFX11 max = 63, GFX12 max = 255 -> both clamp
+
+; GFX11-LABEL: .amdhsa_kernel test_clamping_both
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end7-test_clamping_both, 6)
+; GFX12-LABEL: .amdhsa_kernel test_clamping_both
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end7-test_clamping_both, 8)
+;; Object: kernel descriptor at 0x1C0, COMPUTE_PGM_RSRC3 at 0x1EC:
+;; gfx11: clamped to 63 -> 63<<4 = 0x3F0
+;; gfx12: clamped to 255 -> 255<<4 = 0xFF0
+; OBJ-GFX11: 01e0 {{.*}}f0030000
+; OBJ-GFX12: 01e0 {{.*}}f00f0000
+
+define amdgpu_kernel void @test_clamping_both() {
+ call void asm sideeffect ".fill 8192, 4, 0", ""()
+ ret void
+}
>From dda5b5d353829df9f6f349e9d8f0c933988595e6 Mon Sep 17 00:00:00 2001
From: Adel Ejjeh <adel.ejjeh at amd.com>
Date: Thu, 16 Apr 2026 11:33:17 -0500
Subject: [PATCH 05/11] Refactor inst prefetch size checks
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 22 ++++---------
llvm/lib/Target/AMDGPU/GCNSubtarget.h | 18 +++++++++++
.../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 16 ++++++----
.../Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h | 3 +-
.../test/CodeGen/AMDGPU/inst-prefetch-hint.ll | 21 ++++++------
.../AMDGPU/inst-prefetch-inline-asm.ll | 32 +++++++++----------
6 files changed, 63 insertions(+), 49 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index a6f13b9d35cb2..27696444fcc1f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -266,26 +266,16 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
// compute_pgm_rsrc3, because the label subtraction MCExpr is unresolvable
// in text mode and would prevent printing of other fields (e.g.
// named_barrier_count) that share the same register.
- if (isGFX11Plus(STM)) {
+ if (STM.hasInstPrefSize()) {
const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
- uint32_t Width;
- if (isGFX11(STM)) {
- KD.inst_pref_size_mask = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE;
- KD.inst_pref_size_shift =
- amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT;
- Width = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_WIDTH;
- } else {
- KD.inst_pref_size_mask =
- amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE;
- KD.inst_pref_size_shift =
- amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT;
- Width = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_WIDTH;
- }
- KD.inst_pref_size =
- AMDGPUMCExpr::createInstPrefSize(CodeSizeExpr, Width, Ctx);
+ uint32_t Width, CacheLineSize;
+ STM.getInstPrefSizeArgs(KD.inst_pref_size_mask, KD.inst_pref_size_shift,
+ Width, CacheLineSize);
+ KD.inst_pref_size = AMDGPUMCExpr::createInstPrefSize(CodeSizeExpr, Width,
+ CacheLineSize, Ctx);
}
auto &Streamer = getTargetStreamer()->getStreamer();
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index ec7dd92d6b10e..c66a541b56d4d 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -21,6 +21,7 @@
#include "SIISelLowering.h"
#include "SIInstrInfo.h"
#include "Utils/AMDGPUBaseInfo.h"
+#include "llvm/Support/AMDHSAKernelDescriptor.h"
#include "llvm/Support/ErrorHandling.h"
#define GET_SUBTARGETINFO_HEADER
@@ -429,6 +430,23 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
bool hasPrefetch() const { return HasGFX12Insts; }
+ bool hasInstPrefSize() const { return isGFX11Plus(); }
+
+ void getInstPrefSizeArgs(uint32_t &Mask, uint32_t &Shift, uint32_t &Width,
+ uint32_t &CacheLineSize) const {
+ assert(isGFX11Plus());
+ CacheLineSize = getInstCacheLineSize();
+ if (getGeneration() == GFX11) {
+ Mask = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE;
+ Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT;
+ Width = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_WIDTH;
+ } else {
+ Mask = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE;
+ Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT;
+ Width = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_WIDTH;
+ }
+ }
+
// Has s_cmpk_* instructions.
bool hasSCmpK() const { return getGeneration() < GFX12; }
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index 48c4491f22db3..d0fd807d448d1 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -197,14 +197,15 @@ bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
return true;
};
- assert(Args.size() == 2 &&
+ assert(Args.size() == 3 &&
"AMDGPUMCExpr Argument count incorrect for InstPrefSize");
- uint64_t CodeSizeInBytes = 0, FieldWidth = 0;
+ uint64_t CodeSizeInBytes = 0, FieldWidth = 0, CacheLineSize = 0;
if (!TryGetMCExprValue(Args[0], CodeSizeInBytes) ||
- !TryGetMCExprValue(Args[1], FieldWidth))
+ !TryGetMCExprValue(Args[1], FieldWidth) ||
+ !TryGetMCExprValue(Args[2], CacheLineSize))
return false;
- uint64_t CodeSizeInLines = divideCeil(CodeSizeInBytes, (uint64_t)128);
+ uint64_t CodeSizeInLines = divideCeil(CodeSizeInBytes, CacheLineSize);
uint64_t MaxVal = (1u << FieldWidth) - 1;
Res = MCValue::get(std::min(CodeSizeInLines, MaxVal));
return true;
@@ -311,9 +312,12 @@ const AMDGPUMCExpr *AMDGPUMCExpr::createTotalNumVGPR(const MCExpr *NumAGPR,
const AMDGPUMCExpr *
AMDGPUMCExpr::createInstPrefSize(const MCExpr *CodeSizeBytes,
- unsigned FieldWidth, MCContext &Ctx) {
+ unsigned FieldWidth, unsigned CacheLineSize,
+ MCContext &Ctx) {
return create(AGVK_InstPrefSize,
- {CodeSizeBytes, MCConstantExpr::create(FieldWidth, Ctx)}, Ctx);
+ {CodeSizeBytes, MCConstantExpr::create(FieldWidth, Ctx),
+ MCConstantExpr::create(CacheLineSize, Ctx)},
+ Ctx);
}
const AMDGPUMCExpr *AMDGPUMCExpr::createLit(LitModifier Lit, int64_t Value,
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index 52c984ac450a7..67a4074df4292 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -100,9 +100,10 @@ class AMDGPUMCExpr : public MCTargetExpr {
}
/// Create an expression for instruction prefetch size computation:
- /// min(divideCeil(CodeSizeBytes, 128), (1 << FieldWidth) - 1)
+ /// min(divideCeil(CodeSizeBytes, CacheLineSize), (1 << FieldWidth) - 1)
static const AMDGPUMCExpr *createInstPrefSize(const MCExpr *CodeSizeBytes,
unsigned FieldWidth,
+ unsigned CacheLineSize,
MCContext &Ctx);
static const AMDGPUMCExpr *createLit(LitModifier Lit, int64_t Value,
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
index 99adbce0d37dc..e2e803afa3a22 100644
--- a/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
@@ -11,16 +11,17 @@
; The inst_pref_size is computed via MCExpr label subtraction, resolved at
; assembly/link time. In text output it appears as:
-; instprefsize(<code_size>, <field_width>)
+; instprefsize(<code_size>, <field_width>, <cache_line_size>)
; where:
-; <code_size> = .Lfunc_endN - func_sym (exact function code size in bytes)
-; <field_width> = bit width of the inst_pref_size field (6 for GFX11, 8 for GFX12+)
-; instprefsize = min(divideCeil(code_size, 128), (1 << field_width) - 1)
+; <code_size> = .Lfunc_endN - func_sym (exact function code size in bytes)
+; <field_width> = bit width of the inst_pref_size field (6 for GFX11, 8 for GFX12+)
+; <cache_line_size> = instruction cache line size in bytes (128 for GFX11+)
+; instprefsize = min(divideCeil(code_size, cache_line_size), (1 << field_width) - 1)
; GCN-LABEL: .amdhsa_kernel large
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-large, 6)
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-large, 6, 128)
; GFX11: codeLenInByte = {{[0-9]+}}
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-large, 8)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-large, 8, 128)
; GFX12: codeLenInByte = {{[0-9]+}}
;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C:
;; gfx11 pref=3 (0x30), gfx12 pref=4 (0x40)
@@ -33,8 +34,8 @@ bb:
}
; GCN-LABEL: .amdhsa_kernel small
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-small, 6)
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-small, 8)
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-small, 6, 128)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-small, 8, 128)
; GCN: codeLenInByte = {{[0-9]+}}
;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C:
;; pref=1 (0x10) for both
@@ -49,8 +50,8 @@ bb:
; The MCExpr resolves to the correct inst_pref_size at assembly time.
; GCN-LABEL: .amdhsa_kernel inline_asm
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-inline_asm, 6)
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-inline_asm, 8)
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-inline_asm, 6, 128)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-inline_asm, 8, 128)
; GCN: codeLenInByte = {{[0-9]+}}
;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC:
;; pref=9 (0x90) for both
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
index 1dff24e6c234f..8066ee864627e 100644
--- a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
@@ -14,9 +14,9 @@
;; pref_size = divideCeil(1028, 128) = 9
; GFX11-LABEL: .amdhsa_kernel test_fill
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-test_fill, 6)
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-test_fill, 6, 128)
; GFX12-LABEL: .amdhsa_kernel test_fill
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-test_fill, 8)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-test_fill, 8, 128)
;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C:
;; pref_size=9 -> 9<<4 = 0x90
; OBJ-GFX11: 0020 {{.*}}90000000
@@ -31,9 +31,9 @@ define amdgpu_kernel void @test_fill() {
;; pref_size = 9
; GFX11-LABEL: .amdhsa_kernel test_space
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-test_space, 6)
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-test_space, 6, 128)
; GFX12-LABEL: .amdhsa_kernel test_space
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-test_space, 8)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-test_space, 8, 128)
;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C:
;; pref_size=9 -> 9<<4 = 0x90
; OBJ-GFX11: 0060 {{.*}}90000000
@@ -48,9 +48,9 @@ define amdgpu_kernel void @test_space() {
;; pref_size = divideCeil(132, 128) = 2
; GFX11-LABEL: .amdhsa_kernel test_instructions
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-test_instructions, 6)
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-test_instructions, 6, 128)
; GFX12-LABEL: .amdhsa_kernel test_instructions
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-test_instructions, 8)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-test_instructions, 8, 128)
;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC:
;; pref_size=2 -> 2<<4 = 0x20
; OBJ-GFX11: 00a0 {{.*}}20000000
@@ -65,9 +65,9 @@ define amdgpu_kernel void @test_instructions() {
;; pref_size = 1
; GFX11-LABEL: .amdhsa_kernel test_comments
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end3-test_comments, 6)
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end3-test_comments, 6, 128)
; GFX12-LABEL: .amdhsa_kernel test_comments
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end3-test_comments, 8)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end3-test_comments, 8, 128)
;; Object: kernel descriptor at 0xC0, COMPUTE_PGM_RSRC3 at 0xEC:
;; pref_size=1 -> 1<<4 = 0x10
; OBJ-GFX11: 00e0 {{.*}}10000000
@@ -82,9 +82,9 @@ define amdgpu_kernel void @test_comments() {
;; pref_size = 1
; GFX11-LABEL: .amdhsa_kernel test_empty_asm
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end4-test_empty_asm, 6)
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end4-test_empty_asm, 6, 128)
; GFX12-LABEL: .amdhsa_kernel test_empty_asm
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end4-test_empty_asm, 8)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end4-test_empty_asm, 8, 128)
;; Object: kernel descriptor at 0x100, COMPUTE_PGM_RSRC3 at 0x12C:
;; pref_size=1 -> 1<<4 = 0x10
; OBJ-GFX11: 0120 {{.*}}10000000
@@ -99,9 +99,9 @@ define amdgpu_kernel void @test_empty_asm() {
;; pref_size = divideCeil(1028, 128) = 9
; GFX11-LABEL: .amdhsa_kernel test_multiple_asm
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end5-test_multiple_asm, 6)
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end5-test_multiple_asm, 6, 128)
; GFX12-LABEL: .amdhsa_kernel test_multiple_asm
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end5-test_multiple_asm, 8)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end5-test_multiple_asm, 8, 128)
;; Object: kernel descriptor at 0x140, COMPUTE_PGM_RSRC3 at 0x16C:
;; pref_size=9 -> 9<<4 = 0x90
; OBJ-GFX11: 0160 {{.*}}90000000
@@ -119,9 +119,9 @@ define amdgpu_kernel void @test_multiple_asm() {
;; pref_size should clamp to 63
; GFX11-LABEL: .amdhsa_kernel test_clamping
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end6-test_clamping, 6)
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end6-test_clamping, 6, 128)
; GFX12-LABEL: .amdhsa_kernel test_clamping
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end6-test_clamping, 8)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end6-test_clamping, 8, 128)
;; Object: kernel descriptor at 0x180, COMPUTE_PGM_RSRC3 at 0x1AC:
;; gfx11: clamped to 63 -> 63<<4 = 0x3F0
;; gfx12: no clamping, 65 -> 65<<4 = 0x410
@@ -139,9 +139,9 @@ define amdgpu_kernel void @test_clamping() {
;; GFX11 max = 63, GFX12 max = 255 -> both clamp
; GFX11-LABEL: .amdhsa_kernel test_clamping_both
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end7-test_clamping_both, 6)
+; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end7-test_clamping_both, 6, 128)
; GFX12-LABEL: .amdhsa_kernel test_clamping_both
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end7-test_clamping_both, 8)
+; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end7-test_clamping_both, 8, 128)
;; Object: kernel descriptor at 0x1C0, COMPUTE_PGM_RSRC3 at 0x1EC:
;; gfx11: clamped to 63 -> 63<<4 = 0x3F0
;; gfx12: clamped to 255 -> 255<<4 = 0xFF0
>From 479a0151faf528149e6abadeaf6d754e6af38624 Mon Sep 17 00:00:00 2001
From: Adel Ejjeh <adel.ejjeh at amd.com>
Date: Thu, 23 Apr 2026 10:12:20 -0500
Subject: [PATCH 06/11] OR inst_pref_size directly into compute_pgm_rsrc3
Instead of keeping inst_pref_size as a separate field in MCKernelDescriptor
and special-casing it in text/ELF streamers, OR it directly into
compute_pgm_rsrc3 using setBits (like accum_offset and other fields).
This is made possible by improving KnownBits for AGVK_InstPrefSize to
report upper bits as known zero (the result is clamped to FieldWidth bits).
This allows PrintField (which uses bits_get + fold) to correctly extract
inst_pref_size from the composite rsrc3 MCExpr in text mode, eliminating
the need for special handling.
Addresses review comment from arsenm.
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 15 +++----
.../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 12 ++++++
.../MCTargetDesc/AMDGPUMCKernelDescriptor.h | 8 ----
.../MCTargetDesc/AMDGPUTargetStreamer.cpp | 43 ++++---------------
.../test/CodeGen/AMDGPU/inst-prefetch-hint.ll | 14 +++---
.../AMDGPU/inst-prefetch-inline-asm.ll | 32 +++++++-------
6 files changed, 50 insertions(+), 74 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 27696444fcc1f..a85db906c9c36 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -262,20 +262,17 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
// size. At this point .Lfunc_end has been emitted (by the base AsmPrinter)
// right after the function code, so (Lfunc_end - func_sym) gives the
// exact function code size in bytes.
- // We store it as a separate KD field rather than OR'ing into
- // compute_pgm_rsrc3, because the label subtraction MCExpr is unresolvable
- // in text mode and would prevent printing of other fields (e.g.
- // named_barrier_count) that share the same register.
if (STM.hasInstPrefSize()) {
const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
- uint32_t Width, CacheLineSize;
- STM.getInstPrefSizeArgs(KD.inst_pref_size_mask, KD.inst_pref_size_shift,
- Width, CacheLineSize);
- KD.inst_pref_size = AMDGPUMCExpr::createInstPrefSize(CodeSizeExpr, Width,
- CacheLineSize, Ctx);
+ uint32_t Mask, Shift, Width, CacheLineSize;
+ STM.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
+ const MCExpr *InstPrefSize = AMDGPUMCExpr::createInstPrefSize(
+ CodeSizeExpr, Width, CacheLineSize, Ctx);
+ KD.compute_pgm_rsrc3 =
+ setBits(KD.compute_pgm_rsrc3, InstPrefSize, Mask, Shift, Ctx);
}
auto &Streamer = getTargetStreamer()->getStreamer();
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index d0fd807d448d1..1536b7b8d68e4 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -519,6 +519,18 @@ static void targetOpKnownBitsMapHelper(const MCExpr *Expr, KnownBitsMap &KBM,
KBM[Expr] = KnownBits::makeConstant(APValue);
return;
}
+ if (AGVK->getKind() == AMDGPUMCExpr::VariantKind::AGVK_InstPrefSize) {
+ // The result is clamped to (1 << FieldWidth) - 1, so upper bits are
+ // known zero. FieldWidth is always a constant (Args[1]).
+ int64_t FieldWidth;
+ if (AGVK->getSubExpr(1)->evaluateAsAbsolute(FieldWidth) &&
+ FieldWidth > 0 && static_cast<uint64_t>(FieldWidth) < BitWidth) {
+ KnownBits KB(BitWidth);
+ KB.Zero.setHighBits(BitWidth - FieldWidth);
+ KBM[Expr] = KB;
+ return;
+ }
+ }
KBM[Expr] = KnownBits(BitWidth);
return;
}
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCKernelDescriptor.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCKernelDescriptor.h
index d8bd41cdff5ce..26958ac8b9ee1 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCKernelDescriptor.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCKernelDescriptor.h
@@ -34,14 +34,6 @@ struct MCKernelDescriptor {
const MCExpr *kernel_code_properties = nullptr;
const MCExpr *kernarg_preload = nullptr;
- /// Instruction prefetch size, kept separate from compute_pgm_rsrc3 to avoid
- /// contaminating the register MCExpr with an unresolvable label subtraction
- /// (which would prevent text-mode printing of other fields in the register).
- /// The ELF streamer OR's this into compute_pgm_rsrc3 when emitting bytes.
- const MCExpr *inst_pref_size = nullptr;
- uint32_t inst_pref_size_shift = 0;
- uint32_t inst_pref_size_mask = 0;
-
static MCKernelDescriptor
getDefaultAmdhsaKernelDescriptor(const MCSubtargetInfo *STI, MCContext &Ctx);
// MCExpr for:
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
index d75f99bd5452c..47733494d421b 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
@@ -608,32 +608,16 @@ void AMDGPUTargetAsmStreamer::EmitAmdhsaKernelDescriptor(
".amdhsa_shared_vgpr_count");
}
if (IVersion.Major == 11) {
- if (KD.inst_pref_size) {
- // CodeGen path: print the MCExpr directly (label subtraction).
- OS << "\t\t.amdhsa_inst_pref_size ";
- const MCExpr *New = foldAMDGPUMCExpr(KD.inst_pref_size, getContext());
- printAMDGPUMCExpr(New, OS, MAI);
- OS << '\n';
- } else {
- // MC assembler path: extract from compute_pgm_rsrc3.
- PrintField(KD.compute_pgm_rsrc3,
- amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT,
- amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE,
- ".amdhsa_inst_pref_size");
- }
+ PrintField(KD.compute_pgm_rsrc3,
+ amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT,
+ amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE,
+ ".amdhsa_inst_pref_size");
}
if (IVersion.Major >= 12) {
- if (KD.inst_pref_size) {
- OS << "\t\t.amdhsa_inst_pref_size ";
- const MCExpr *New = foldAMDGPUMCExpr(KD.inst_pref_size, getContext());
- printAMDGPUMCExpr(New, OS, MAI);
- OS << '\n';
- } else {
- PrintField(KD.compute_pgm_rsrc3,
- amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT,
- amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE,
- ".amdhsa_inst_pref_size");
- }
+ PrintField(KD.compute_pgm_rsrc3,
+ amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT,
+ amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE,
+ ".amdhsa_inst_pref_size");
PrintField(KD.compute_pgm_rsrc1,
amdhsa::COMPUTE_PGM_RSRC1_GFX12_PLUS_ENABLE_WG_RR_EN_SHIFT,
amdhsa::COMPUTE_PGM_RSRC1_GFX12_PLUS_ENABLE_WG_RR_EN,
@@ -1067,16 +1051,7 @@ void AMDGPUTargetELFStreamer::EmitAmdhsaKernelDescriptor(
sizeof(amdhsa::kernel_descriptor_t::kernel_code_entry_byte_offset));
for (uint32_t i = 0; i < sizeof(amdhsa::kernel_descriptor_t::reserved1); ++i)
Streamer.emitInt8(0u);
- // OR inst_pref_size into compute_pgm_rsrc3 for the binary encoding.
- // This is kept separate in the KD struct to avoid making the MCExpr
- // unresolvable in text mode (see AMDGPUAsmPrinter::endFunction).
- const MCExpr *Rsrc3 = KernelDescriptor.compute_pgm_rsrc3;
- if (KernelDescriptor.inst_pref_size) {
- MCKernelDescriptor::bits_set(Rsrc3, KernelDescriptor.inst_pref_size,
- KernelDescriptor.inst_pref_size_shift,
- KernelDescriptor.inst_pref_size_mask, Context);
- }
- Streamer.emitValue(Rsrc3,
+ Streamer.emitValue(KernelDescriptor.compute_pgm_rsrc3,
sizeof(amdhsa::kernel_descriptor_t::compute_pgm_rsrc3));
Streamer.emitValue(KernelDescriptor.compute_pgm_rsrc1,
sizeof(amdhsa::kernel_descriptor_t::compute_pgm_rsrc1));
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
index e2e803afa3a22..0b9eebbe52644 100644
--- a/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
@@ -11,7 +11,7 @@
; The inst_pref_size is computed via MCExpr label subtraction, resolved at
; assembly/link time. In text output it appears as:
-; instprefsize(<code_size>, <field_width>, <cache_line_size>)
+; ((instprefsize(<code_size>, <field_width>, <cache_line_size>)<<Shift)&Mask)>>Shift
; where:
; <code_size> = .Lfunc_endN - func_sym (exact function code size in bytes)
; <field_width> = bit width of the inst_pref_size field (6 for GFX11, 8 for GFX12+)
@@ -19,9 +19,9 @@
; instprefsize = min(divideCeil(code_size, cache_line_size), (1 << field_width) - 1)
; GCN-LABEL: .amdhsa_kernel large
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-large, 6, 128)
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-large, 6, 128)<<4)&1008)>>4
; GFX11: codeLenInByte = {{[0-9]+}}
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-large, 8, 128)
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-large, 8, 128)<<4)&4080)>>4
; GFX12: codeLenInByte = {{[0-9]+}}
;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C:
;; gfx11 pref=3 (0x30), gfx12 pref=4 (0x40)
@@ -34,8 +34,8 @@ bb:
}
; GCN-LABEL: .amdhsa_kernel small
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-small, 6, 128)
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-small, 8, 128)
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-small, 6, 128)<<4)&1008)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-small, 8, 128)<<4)&4080)>>4
; GCN: codeLenInByte = {{[0-9]+}}
;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C:
;; pref=1 (0x10) for both
@@ -50,8 +50,8 @@ bb:
; The MCExpr resolves to the correct inst_pref_size at assembly time.
; GCN-LABEL: .amdhsa_kernel inline_asm
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-inline_asm, 6, 128)
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-inline_asm, 8, 128)
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-inline_asm, 6, 128)<<4)&1008)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-inline_asm, 8, 128)<<4)&4080)>>4
; GCN: codeLenInByte = {{[0-9]+}}
;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC:
;; pref=9 (0x90) for both
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
index 8066ee864627e..c1ca30e8243c8 100644
--- a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
@@ -14,9 +14,9 @@
;; pref_size = divideCeil(1028, 128) = 9
; GFX11-LABEL: .amdhsa_kernel test_fill
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-test_fill, 6, 128)
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-test_fill, 6, 128)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_fill
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end0-test_fill, 8, 128)
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-test_fill, 8, 128)<<4)&4080)>>4
;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C:
;; pref_size=9 -> 9<<4 = 0x90
; OBJ-GFX11: 0020 {{.*}}90000000
@@ -31,9 +31,9 @@ define amdgpu_kernel void @test_fill() {
;; pref_size = 9
; GFX11-LABEL: .amdhsa_kernel test_space
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-test_space, 6, 128)
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-test_space, 6, 128)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_space
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end1-test_space, 8, 128)
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-test_space, 8, 128)<<4)&4080)>>4
;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C:
;; pref_size=9 -> 9<<4 = 0x90
; OBJ-GFX11: 0060 {{.*}}90000000
@@ -48,9 +48,9 @@ define amdgpu_kernel void @test_space() {
;; pref_size = divideCeil(132, 128) = 2
; GFX11-LABEL: .amdhsa_kernel test_instructions
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-test_instructions, 6, 128)
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-test_instructions, 6, 128)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_instructions
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end2-test_instructions, 8, 128)
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-test_instructions, 8, 128)<<4)&4080)>>4
;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC:
;; pref_size=2 -> 2<<4 = 0x20
; OBJ-GFX11: 00a0 {{.*}}20000000
@@ -65,9 +65,9 @@ define amdgpu_kernel void @test_instructions() {
;; pref_size = 1
; GFX11-LABEL: .amdhsa_kernel test_comments
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end3-test_comments, 6, 128)
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end3-test_comments, 6, 128)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_comments
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end3-test_comments, 8, 128)
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end3-test_comments, 8, 128)<<4)&4080)>>4
;; Object: kernel descriptor at 0xC0, COMPUTE_PGM_RSRC3 at 0xEC:
;; pref_size=1 -> 1<<4 = 0x10
; OBJ-GFX11: 00e0 {{.*}}10000000
@@ -82,9 +82,9 @@ define amdgpu_kernel void @test_comments() {
;; pref_size = 1
; GFX11-LABEL: .amdhsa_kernel test_empty_asm
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end4-test_empty_asm, 6, 128)
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end4-test_empty_asm, 6, 128)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_empty_asm
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end4-test_empty_asm, 8, 128)
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end4-test_empty_asm, 8, 128)<<4)&4080)>>4
;; Object: kernel descriptor at 0x100, COMPUTE_PGM_RSRC3 at 0x12C:
;; pref_size=1 -> 1<<4 = 0x10
; OBJ-GFX11: 0120 {{.*}}10000000
@@ -99,9 +99,9 @@ define amdgpu_kernel void @test_empty_asm() {
;; pref_size = divideCeil(1028, 128) = 9
; GFX11-LABEL: .amdhsa_kernel test_multiple_asm
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end5-test_multiple_asm, 6, 128)
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end5-test_multiple_asm, 6, 128)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_multiple_asm
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end5-test_multiple_asm, 8, 128)
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end5-test_multiple_asm, 8, 128)<<4)&4080)>>4
;; Object: kernel descriptor at 0x140, COMPUTE_PGM_RSRC3 at 0x16C:
;; pref_size=9 -> 9<<4 = 0x90
; OBJ-GFX11: 0160 {{.*}}90000000
@@ -119,9 +119,9 @@ define amdgpu_kernel void @test_multiple_asm() {
;; pref_size should clamp to 63
; GFX11-LABEL: .amdhsa_kernel test_clamping
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end6-test_clamping, 6, 128)
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end6-test_clamping, 6, 128)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_clamping
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end6-test_clamping, 8, 128)
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end6-test_clamping, 8, 128)<<4)&4080)>>4
;; Object: kernel descriptor at 0x180, COMPUTE_PGM_RSRC3 at 0x1AC:
;; gfx11: clamped to 63 -> 63<<4 = 0x3F0
;; gfx12: no clamping, 65 -> 65<<4 = 0x410
@@ -139,9 +139,9 @@ define amdgpu_kernel void @test_clamping() {
;; GFX11 max = 63, GFX12 max = 255 -> both clamp
; GFX11-LABEL: .amdhsa_kernel test_clamping_both
-; GFX11: .amdhsa_inst_pref_size instprefsize(.Lfunc_end7-test_clamping_both, 6, 128)
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end7-test_clamping_both, 6, 128)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_clamping_both
-; GFX12: .amdhsa_inst_pref_size instprefsize(.Lfunc_end7-test_clamping_both, 8, 128)
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end7-test_clamping_both, 8, 128)<<4)&4080)>>4
;; Object: kernel descriptor at 0x1C0, COMPUTE_PGM_RSRC3 at 0x1EC:
;; gfx11: clamped to 63 -> 63<<4 = 0x3F0
;; gfx12: clamped to 255 -> 255<<4 = 0xFF0
>From 7cdb1e898f4bcbf297cc25bd220de617f4fa8fa1 Mon Sep 17 00:00:00 2001
From: Adel Ejjeh <adel.ejjeh at amd.com>
Date: Thu, 23 Apr 2026 16:33:12 -0500
Subject: [PATCH 07/11] Derive inst_pref_size FieldWidth and CacheLineSize from
STI
Instead of passing FieldWidth and CacheLineSize as MCExpr arguments
to createInstPrefSize, derive them from the MCSubtargetInfo inside
evaluateInstPrefSize and the KnownBits helper. This simplifies the
MCExpr to a single argument (CodeSizeBytes) and reduces the printed
expression from instprefsize(X, W, C) to instprefsize(X).
All GFX11+ targets use 128-byte instruction cache lines. The field
width (6 for GFX11, 8 for GFX12+) is looked up via getIsaVersion.
Also updated evaluateInstPrefSize to use the new evaluateMCExprs
initializer_list API after rebasing on top of the merged NFC refactor.
Addresses review comment from jayfoad.
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 4 +-
.../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 51 +++++++++----------
.../Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h | 4 +-
.../test/CodeGen/AMDGPU/inst-prefetch-hint.ll | 17 +++----
.../AMDGPU/inst-prefetch-inline-asm.ll | 32 ++++++------
5 files changed, 53 insertions(+), 55 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index a85db906c9c36..365e63e738ab4 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -269,8 +269,8 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
uint32_t Mask, Shift, Width, CacheLineSize;
STM.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
- const MCExpr *InstPrefSize = AMDGPUMCExpr::createInstPrefSize(
- CodeSizeExpr, Width, CacheLineSize, Ctx);
+ const MCExpr *InstPrefSize =
+ AMDGPUMCExpr::createInstPrefSize(CodeSizeExpr, Ctx);
KD.compute_pgm_rsrc3 =
setBits(KD.compute_pgm_rsrc3, InstPrefSize, Mask, Shift, Ctx);
}
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index 1536b7b8d68e4..80f948682ab4a 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -12,11 +12,14 @@
#include "llvm/MC/MCAssembler.h"
#include "llvm/MC/MCContext.h"
#include "llvm/MC/MCStreamer.h"
+#include "llvm/MC/MCSubtargetInfo.h"
#include "llvm/MC/MCSymbol.h"
#include "llvm/MC/MCValue.h"
+#include "llvm/Support/AMDHSAKernelDescriptor.h"
#include "llvm/Support/KnownBits.h"
#include "llvm/Support/MathExtras.h"
#include "llvm/Support/raw_ostream.h"
+#include "llvm/TargetParser/TargetParser.h"
#include <functional>
#include <optional>
@@ -186,25 +189,27 @@ bool AMDGPUMCExpr::evaluateOccupancy(MCValue &Res,
return true;
}
+/// Get the inst_pref_size field width for the given subtarget.
+static unsigned getInstPrefSizeFieldWidth(const MCSubtargetInfo *STI) {
+ auto Version = getIsaVersion(STI->getCPU());
+ assert(Version.Major >= 11 && "inst_pref_size only exists on GFX11+");
+ if (Version.Major == 11)
+ return amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_WIDTH;
+ return amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_WIDTH;
+}
+
bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
const MCAssembler *Asm) const {
- auto TryGetMCExprValue = [&](const MCExpr *Arg, uint64_t &ConstantValue) {
- MCValue MCVal;
- if (!Arg->evaluateAsRelocatable(MCVal, Asm) || !MCVal.isAbsolute())
- return false;
-
- ConstantValue = MCVal.getConstant();
- return true;
- };
-
- assert(Args.size() == 3 &&
+ assert(Args.size() == 1 &&
"AMDGPUMCExpr Argument count incorrect for InstPrefSize");
- uint64_t CodeSizeInBytes = 0, FieldWidth = 0, CacheLineSize = 0;
- if (!TryGetMCExprValue(Args[0], CodeSizeInBytes) ||
- !TryGetMCExprValue(Args[1], FieldWidth) ||
- !TryGetMCExprValue(Args[2], CacheLineSize))
- return false;
+ uint64_t CodeSizeInBytes = 0;
+ if (!evaluateMCExprs(Args, Asm, {CodeSizeInBytes}))
+ return false;
+ const MCSubtargetInfo *STI = Ctx.getSubtargetInfo();
+ unsigned FieldWidth = getInstPrefSizeFieldWidth(STI);
+ // All targets with inst_pref_size (GFX11+) have 128-byte cache lines.
+ constexpr unsigned CacheLineSize = 128;
uint64_t CodeSizeInLines = divideCeil(CodeSizeInBytes, CacheLineSize);
uint64_t MaxVal = (1u << FieldWidth) - 1;
Res = MCValue::get(std::min(CodeSizeInLines, MaxVal));
@@ -311,13 +316,8 @@ const AMDGPUMCExpr *AMDGPUMCExpr::createTotalNumVGPR(const MCExpr *NumAGPR,
}
const AMDGPUMCExpr *
-AMDGPUMCExpr::createInstPrefSize(const MCExpr *CodeSizeBytes,
- unsigned FieldWidth, unsigned CacheLineSize,
- MCContext &Ctx) {
- return create(AGVK_InstPrefSize,
- {CodeSizeBytes, MCConstantExpr::create(FieldWidth, Ctx),
- MCConstantExpr::create(CacheLineSize, Ctx)},
- Ctx);
+AMDGPUMCExpr::createInstPrefSize(const MCExpr *CodeSizeBytes, MCContext &Ctx) {
+ return create(AGVK_InstPrefSize, {CodeSizeBytes}, Ctx);
}
const AMDGPUMCExpr *AMDGPUMCExpr::createLit(LitModifier Lit, int64_t Value,
@@ -521,10 +521,9 @@ static void targetOpKnownBitsMapHelper(const MCExpr *Expr, KnownBitsMap &KBM,
}
if (AGVK->getKind() == AMDGPUMCExpr::VariantKind::AGVK_InstPrefSize) {
// The result is clamped to (1 << FieldWidth) - 1, so upper bits are
- // known zero. FieldWidth is always a constant (Args[1]).
- int64_t FieldWidth;
- if (AGVK->getSubExpr(1)->evaluateAsAbsolute(FieldWidth) &&
- FieldWidth > 0 && static_cast<uint64_t>(FieldWidth) < BitWidth) {
+ // known zero. FieldWidth is derived from the subtarget.
+ if (const MCSubtargetInfo *STI = AGVK->getCtx().getSubtargetInfo()) {
+ unsigned FieldWidth = getInstPrefSizeFieldWidth(STI);
KnownBits KB(BitWidth);
KB.Zero.setHighBits(BitWidth - FieldWidth);
KBM[Expr] = KB;
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index 67a4074df4292..4b1aa0c591a80 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -101,9 +101,8 @@ class AMDGPUMCExpr : public MCTargetExpr {
/// Create an expression for instruction prefetch size computation:
/// min(divideCeil(CodeSizeBytes, CacheLineSize), (1 << FieldWidth) - 1)
+ /// FieldWidth and CacheLineSize are derived from the subtarget.
static const AMDGPUMCExpr *createInstPrefSize(const MCExpr *CodeSizeBytes,
- unsigned FieldWidth,
- unsigned CacheLineSize,
MCContext &Ctx);
static const AMDGPUMCExpr *createLit(LitModifier Lit, int64_t Value,
@@ -111,6 +110,7 @@ class AMDGPUMCExpr : public MCTargetExpr {
ArrayRef<const MCExpr *> getArgs() const { return Args; }
VariantKind getKind() const { return Kind; }
+ MCContext &getCtx() const { return Ctx; }
const MCExpr *getSubExpr(size_t Index) const;
void printImpl(raw_ostream &OS, const MCAsmInfo *MAI) const override;
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
index 0b9eebbe52644..b76ef7eac11c4 100644
--- a/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-hint.ll
@@ -11,17 +11,16 @@
; The inst_pref_size is computed via MCExpr label subtraction, resolved at
; assembly/link time. In text output it appears as:
-; ((instprefsize(<code_size>, <field_width>, <cache_line_size>)<<Shift)&Mask)>>Shift
+; ((instprefsize(<code_size>)<<Shift)&Mask)>>Shift
; where:
; <code_size> = .Lfunc_endN - func_sym (exact function code size in bytes)
-; <field_width> = bit width of the inst_pref_size field (6 for GFX11, 8 for GFX12+)
-; <cache_line_size> = instruction cache line size in bytes (128 for GFX11+)
; instprefsize = min(divideCeil(code_size, cache_line_size), (1 << field_width) - 1)
+; field_width and cache_line_size are derived from the subtarget
; GCN-LABEL: .amdhsa_kernel large
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-large, 6, 128)<<4)&1008)>>4
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-large)<<4)&1008)>>4
; GFX11: codeLenInByte = {{[0-9]+}}
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-large, 8, 128)<<4)&4080)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-large)<<4)&4080)>>4
; GFX12: codeLenInByte = {{[0-9]+}}
;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C:
;; gfx11 pref=3 (0x30), gfx12 pref=4 (0x40)
@@ -34,8 +33,8 @@ bb:
}
; GCN-LABEL: .amdhsa_kernel small
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-small, 6, 128)<<4)&1008)>>4
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-small, 8, 128)<<4)&4080)>>4
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-small)<<4)&1008)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-small)<<4)&4080)>>4
; GCN: codeLenInByte = {{[0-9]+}}
;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C:
;; pref=1 (0x10) for both
@@ -50,8 +49,8 @@ bb:
; The MCExpr resolves to the correct inst_pref_size at assembly time.
; GCN-LABEL: .amdhsa_kernel inline_asm
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-inline_asm, 6, 128)<<4)&1008)>>4
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-inline_asm, 8, 128)<<4)&4080)>>4
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-inline_asm)<<4)&1008)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-inline_asm)<<4)&4080)>>4
; GCN: codeLenInByte = {{[0-9]+}}
;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC:
;; pref=9 (0x90) for both
diff --git a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
index c1ca30e8243c8..287a30032230b 100644
--- a/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
+++ b/llvm/test/CodeGen/AMDGPU/inst-prefetch-inline-asm.ll
@@ -14,9 +14,9 @@
;; pref_size = divideCeil(1028, 128) = 9
; GFX11-LABEL: .amdhsa_kernel test_fill
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-test_fill, 6, 128)<<4)&1008)>>4
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-test_fill)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_fill
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-test_fill, 8, 128)<<4)&4080)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end0-test_fill)<<4)&4080)>>4
;; Object: kernel descriptor at 0x00, COMPUTE_PGM_RSRC3 at 0x2C:
;; pref_size=9 -> 9<<4 = 0x90
; OBJ-GFX11: 0020 {{.*}}90000000
@@ -31,9 +31,9 @@ define amdgpu_kernel void @test_fill() {
;; pref_size = 9
; GFX11-LABEL: .amdhsa_kernel test_space
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-test_space, 6, 128)<<4)&1008)>>4
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-test_space)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_space
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-test_space, 8, 128)<<4)&4080)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end1-test_space)<<4)&4080)>>4
;; Object: kernel descriptor at 0x40, COMPUTE_PGM_RSRC3 at 0x6C:
;; pref_size=9 -> 9<<4 = 0x90
; OBJ-GFX11: 0060 {{.*}}90000000
@@ -48,9 +48,9 @@ define amdgpu_kernel void @test_space() {
;; pref_size = divideCeil(132, 128) = 2
; GFX11-LABEL: .amdhsa_kernel test_instructions
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-test_instructions, 6, 128)<<4)&1008)>>4
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-test_instructions)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_instructions
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-test_instructions, 8, 128)<<4)&4080)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end2-test_instructions)<<4)&4080)>>4
;; Object: kernel descriptor at 0x80, COMPUTE_PGM_RSRC3 at 0xAC:
;; pref_size=2 -> 2<<4 = 0x20
; OBJ-GFX11: 00a0 {{.*}}20000000
@@ -65,9 +65,9 @@ define amdgpu_kernel void @test_instructions() {
;; pref_size = 1
; GFX11-LABEL: .amdhsa_kernel test_comments
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end3-test_comments, 6, 128)<<4)&1008)>>4
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end3-test_comments)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_comments
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end3-test_comments, 8, 128)<<4)&4080)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end3-test_comments)<<4)&4080)>>4
;; Object: kernel descriptor at 0xC0, COMPUTE_PGM_RSRC3 at 0xEC:
;; pref_size=1 -> 1<<4 = 0x10
; OBJ-GFX11: 00e0 {{.*}}10000000
@@ -82,9 +82,9 @@ define amdgpu_kernel void @test_comments() {
;; pref_size = 1
; GFX11-LABEL: .amdhsa_kernel test_empty_asm
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end4-test_empty_asm, 6, 128)<<4)&1008)>>4
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end4-test_empty_asm)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_empty_asm
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end4-test_empty_asm, 8, 128)<<4)&4080)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end4-test_empty_asm)<<4)&4080)>>4
;; Object: kernel descriptor at 0x100, COMPUTE_PGM_RSRC3 at 0x12C:
;; pref_size=1 -> 1<<4 = 0x10
; OBJ-GFX11: 0120 {{.*}}10000000
@@ -99,9 +99,9 @@ define amdgpu_kernel void @test_empty_asm() {
;; pref_size = divideCeil(1028, 128) = 9
; GFX11-LABEL: .amdhsa_kernel test_multiple_asm
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end5-test_multiple_asm, 6, 128)<<4)&1008)>>4
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end5-test_multiple_asm)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_multiple_asm
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end5-test_multiple_asm, 8, 128)<<4)&4080)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end5-test_multiple_asm)<<4)&4080)>>4
;; Object: kernel descriptor at 0x140, COMPUTE_PGM_RSRC3 at 0x16C:
;; pref_size=9 -> 9<<4 = 0x90
; OBJ-GFX11: 0160 {{.*}}90000000
@@ -119,9 +119,9 @@ define amdgpu_kernel void @test_multiple_asm() {
;; pref_size should clamp to 63
; GFX11-LABEL: .amdhsa_kernel test_clamping
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end6-test_clamping, 6, 128)<<4)&1008)>>4
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end6-test_clamping)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_clamping
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end6-test_clamping, 8, 128)<<4)&4080)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end6-test_clamping)<<4)&4080)>>4
;; Object: kernel descriptor at 0x180, COMPUTE_PGM_RSRC3 at 0x1AC:
;; gfx11: clamped to 63 -> 63<<4 = 0x3F0
;; gfx12: no clamping, 65 -> 65<<4 = 0x410
@@ -139,9 +139,9 @@ define amdgpu_kernel void @test_clamping() {
;; GFX11 max = 63, GFX12 max = 255 -> both clamp
; GFX11-LABEL: .amdhsa_kernel test_clamping_both
-; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end7-test_clamping_both, 6, 128)<<4)&1008)>>4
+; GFX11: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end7-test_clamping_both)<<4)&1008)>>4
; GFX12-LABEL: .amdhsa_kernel test_clamping_both
-; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end7-test_clamping_both, 8, 128)<<4)&4080)>>4
+; GFX12: .amdhsa_inst_pref_size ((instprefsize(.Lfunc_end7-test_clamping_both)<<4)&4080)>>4
;; Object: kernel descriptor at 0x1C0, COMPUTE_PGM_RSRC3 at 0x1EC:
;; gfx11: clamped to 63 -> 63<<4 = 0x3F0
;; gfx12: clamped to 255 -> 255<<4 = 0xFF0
>From fab87a06294073090b60a31db9cd1b56ecf1b5c9 Mon Sep 17 00:00:00 2001
From: Adel Ejjeh <adel.ejjeh at amd.com>
Date: Mon, 27 Apr 2026 10:16:04 -0500
Subject: [PATCH 08/11] Update handing of cachelinesize in amdgpumcexpr and
replace setHighBits with setBitsFrom
---
llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 8 ++------
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 8 ++++++++
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 3 +++
3 files changed, 13 insertions(+), 6 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index 80f948682ab4a..c27e4d727db27 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -200,16 +200,12 @@ static unsigned getInstPrefSizeFieldWidth(const MCSubtargetInfo *STI) {
bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
const MCAssembler *Asm) const {
- assert(Args.size() == 1 &&
- "AMDGPUMCExpr Argument count incorrect for InstPrefSize");
-
uint64_t CodeSizeInBytes = 0;
if (!evaluateMCExprs(Args, Asm, {CodeSizeInBytes}))
return false;
const MCSubtargetInfo *STI = Ctx.getSubtargetInfo();
unsigned FieldWidth = getInstPrefSizeFieldWidth(STI);
- // All targets with inst_pref_size (GFX11+) have 128-byte cache lines.
- constexpr unsigned CacheLineSize = 128;
+ unsigned CacheLineSize = AMDGPU::IsaInfo::getInstCacheLineSize(STI);
uint64_t CodeSizeInLines = divideCeil(CodeSizeInBytes, CacheLineSize);
uint64_t MaxVal = (1u << FieldWidth) - 1;
Res = MCValue::get(std::min(CodeSizeInLines, MaxVal));
@@ -525,7 +521,7 @@ static void targetOpKnownBitsMapHelper(const MCExpr *Expr, KnownBitsMap &KBM,
if (const MCSubtargetInfo *STI = AGVK->getCtx().getSubtargetInfo()) {
unsigned FieldWidth = getInstPrefSizeFieldWidth(STI);
KnownBits KB(BitWidth);
- KB.Zero.setHighBits(BitWidth - FieldWidth);
+ KB.Zero.setBitsFrom(FieldWidth);
KBM[Expr] = KB;
return;
}
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 1c145359ccc61..91f9d9e6c9434 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1177,6 +1177,14 @@ std::string AMDGPUTargetID::toString() const {
return Str;
}
+unsigned getInstCacheLineSize(const MCSubtargetInfo *STI) {
+ if (STI->getFeatureBits().test(FeatureInstCacheLineSize128))
+ return 128;
+ if (STI->getFeatureBits().test(FeatureInstCacheLineSize64))
+ return 64;
+ return 64;
+}
+
unsigned getWavefrontSize(const MCSubtargetInfo *STI) {
if (STI->getFeatureBits().test(FeatureWavefrontSize16))
return 16;
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index f6b86a59b7b1d..db88200c937c5 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -232,6 +232,9 @@ inline raw_ostream &operator<<(raw_ostream &OS,
return OS;
}
+/// \returns Instruction cache line size in bytes for given subtarget \p STI.
+unsigned getInstCacheLineSize(const MCSubtargetInfo *STI);
+
/// \returns Wavefront size for given subtarget \p STI.
unsigned getWavefrontSize(const MCSubtargetInfo *STI);
>From 7a3045c8fde128459fc60ec6d4ae9657b74e1745 Mon Sep 17 00:00:00 2001
From: Adel Ejjeh <adel.ejjeh at amd.com>
Date: Mon, 27 Apr 2026 17:26:18 -0500
Subject: [PATCH 09/11] Revert removal of assertions, and fix handing of
generation check and const reference.
---
.../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 23 +++++++++++--------
1 file changed, 14 insertions(+), 9 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index c27e4d727db27..20376c3c1362d 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -19,7 +19,6 @@
#include "llvm/Support/KnownBits.h"
#include "llvm/Support/MathExtras.h"
#include "llvm/Support/raw_ostream.h"
-#include "llvm/TargetParser/TargetParser.h"
#include <functional>
#include <optional>
@@ -123,6 +122,8 @@ evaluateMCExprs(ArrayRef<const MCExpr *> Exprs, const MCAssembler *Asm,
bool AMDGPUMCExpr::evaluateExtraSGPRs(MCValue &Res,
const MCAssembler *Asm) const {
+ assert(Args.size() == 3 &&
+ "AMDGPUMCExpr Argument count incorrect for ExtraSGPRs");
const MCSubtargetInfo *STI = Ctx.getSubtargetInfo();
uint64_t VCCUsed = 0, FlatScrUsed = 0, XNACKUsed = 0;
@@ -137,6 +138,8 @@ bool AMDGPUMCExpr::evaluateExtraSGPRs(MCValue &Res,
bool AMDGPUMCExpr::evaluateTotalNumVGPR(MCValue &Res,
const MCAssembler *Asm) const {
+ assert(Args.size() == 2 &&
+ "AMDGPUMCExpr Argument count incorrect for TotalNumVGPRs");
const MCSubtargetInfo *STI = Ctx.getSubtargetInfo();
uint64_t NumAGPR = 0, NumVGPR = 0;
@@ -152,6 +155,8 @@ bool AMDGPUMCExpr::evaluateTotalNumVGPR(MCValue &Res,
}
bool AMDGPUMCExpr::evaluateAlignTo(MCValue &Res, const MCAssembler *Asm) const {
+ assert(Args.size() == 2 &&
+ "AMDGPUMCExpr Argument count incorrect for AlignTo");
uint64_t Value = 0, Align = 0;
if (!evaluateMCExprs(Args, Asm, {Value, Align}))
return false;
@@ -162,6 +167,8 @@ bool AMDGPUMCExpr::evaluateAlignTo(MCValue &Res, const MCAssembler *Asm) const {
bool AMDGPUMCExpr::evaluateOccupancy(MCValue &Res,
const MCAssembler *Asm) const {
+ assert(Args.size() == 7 &&
+ "AMDGPUMCExpr Argument count incorrect for Occupancy");
uint64_t InitOccupancy, MaxWaves, Granule, TargetTotalNumVGPRs, Generation,
NumSGPRs, NumVGPRs;
@@ -190,12 +197,10 @@ bool AMDGPUMCExpr::evaluateOccupancy(MCValue &Res,
}
/// Get the inst_pref_size field width for the given subtarget.
-static unsigned getInstPrefSizeFieldWidth(const MCSubtargetInfo *STI) {
- auto Version = getIsaVersion(STI->getCPU());
- assert(Version.Major >= 11 && "inst_pref_size only exists on GFX11+");
- if (Version.Major == 11)
- return amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_WIDTH;
- return amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_WIDTH;
+static unsigned getInstPrefSizeFieldWidth(const MCSubtargetInfo &STI) {
+ if (AMDGPU::isGFX12Plus(STI))
+ return amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_WIDTH;
+ return amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_WIDTH;
}
bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
@@ -204,7 +209,7 @@ bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
if (!evaluateMCExprs(Args, Asm, {CodeSizeInBytes}))
return false;
const MCSubtargetInfo *STI = Ctx.getSubtargetInfo();
- unsigned FieldWidth = getInstPrefSizeFieldWidth(STI);
+ unsigned FieldWidth = getInstPrefSizeFieldWidth(*STI);
unsigned CacheLineSize = AMDGPU::IsaInfo::getInstCacheLineSize(STI);
uint64_t CodeSizeInLines = divideCeil(CodeSizeInBytes, CacheLineSize);
uint64_t MaxVal = (1u << FieldWidth) - 1;
@@ -519,7 +524,7 @@ static void targetOpKnownBitsMapHelper(const MCExpr *Expr, KnownBitsMap &KBM,
// The result is clamped to (1 << FieldWidth) - 1, so upper bits are
// known zero. FieldWidth is derived from the subtarget.
if (const MCSubtargetInfo *STI = AGVK->getCtx().getSubtargetInfo()) {
- unsigned FieldWidth = getInstPrefSizeFieldWidth(STI);
+ unsigned FieldWidth = getInstPrefSizeFieldWidth(*STI);
KnownBits KB(BitWidth);
KB.Zero.setBitsFrom(FieldWidth);
KBM[Expr] = KB;
>From e38ba3e25d9b7966a8832e0377b6bba5cd0a20e0 Mon Sep 17 00:00:00 2001
From: Adel Ejjeh <adel.ejjeh at amd.com>
Date: Tue, 28 Apr 2026 09:45:27 -0500
Subject: [PATCH 10/11] Remove redundant assertions after rebase onto main
The NFC patch (PR #194488) that removed Args.size() assertions
landed in main. Remove the re-added assertions from this branch
to align with upstream.
---
llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 8 --------
1 file changed, 8 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index 20376c3c1362d..e968af1d18a93 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -122,8 +122,6 @@ evaluateMCExprs(ArrayRef<const MCExpr *> Exprs, const MCAssembler *Asm,
bool AMDGPUMCExpr::evaluateExtraSGPRs(MCValue &Res,
const MCAssembler *Asm) const {
- assert(Args.size() == 3 &&
- "AMDGPUMCExpr Argument count incorrect for ExtraSGPRs");
const MCSubtargetInfo *STI = Ctx.getSubtargetInfo();
uint64_t VCCUsed = 0, FlatScrUsed = 0, XNACKUsed = 0;
@@ -138,8 +136,6 @@ bool AMDGPUMCExpr::evaluateExtraSGPRs(MCValue &Res,
bool AMDGPUMCExpr::evaluateTotalNumVGPR(MCValue &Res,
const MCAssembler *Asm) const {
- assert(Args.size() == 2 &&
- "AMDGPUMCExpr Argument count incorrect for TotalNumVGPRs");
const MCSubtargetInfo *STI = Ctx.getSubtargetInfo();
uint64_t NumAGPR = 0, NumVGPR = 0;
@@ -155,8 +151,6 @@ bool AMDGPUMCExpr::evaluateTotalNumVGPR(MCValue &Res,
}
bool AMDGPUMCExpr::evaluateAlignTo(MCValue &Res, const MCAssembler *Asm) const {
- assert(Args.size() == 2 &&
- "AMDGPUMCExpr Argument count incorrect for AlignTo");
uint64_t Value = 0, Align = 0;
if (!evaluateMCExprs(Args, Asm, {Value, Align}))
return false;
@@ -167,8 +161,6 @@ bool AMDGPUMCExpr::evaluateAlignTo(MCValue &Res, const MCAssembler *Asm) const {
bool AMDGPUMCExpr::evaluateOccupancy(MCValue &Res,
const MCAssembler *Asm) const {
- assert(Args.size() == 7 &&
- "AMDGPUMCExpr Argument count incorrect for Occupancy");
uint64_t InitOccupancy, MaxWaves, Granule, TargetTotalNumVGPRs, Generation,
NumSGPRs, NumVGPRs;
>From 1c6bad91c4fd0be61dcd1a8aa666e2b8be8ebd5d Mon Sep 17 00:00:00 2001
From: Adel Ejjeh <adel.ejjeh at amd.com>
Date: Wed, 29 Apr 2026 10:59:19 -0500
Subject: [PATCH 11/11] Change createNot(Msk) to create(~Mask)
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 3 +--
1 file changed, 1 insertion(+), 2 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 365e63e738ab4..5d08aef74cbfd 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -238,8 +238,7 @@ void AMDGPUAsmPrinter::emitFunctionBodyStart() {
static const MCExpr *setBits(const MCExpr *Dst, const MCExpr *Value,
uint32_t Mask, uint32_t Shift, MCContext &Ctx) {
const auto *Shft = MCConstantExpr::create(Shift, Ctx);
- const auto *Msk = MCConstantExpr::create(Mask, Ctx);
- Dst = MCBinaryExpr::createAnd(Dst, MCUnaryExpr::createNot(Msk, Ctx), Ctx);
+ Dst = MCBinaryExpr::createAnd(Dst, MCConstantExpr::create(~Mask, Ctx), Ctx);
Dst = MCBinaryExpr::createOr(Dst, MCBinaryExpr::createShl(Value, Shft, Ctx),
Ctx);
return Dst;
More information about the llvm-commits
mailing list