[llvm] [AMDGPU] Explicit instruction prefetch for large entry functions (PR #227995)
Lukas Sommer via llvm-commits
llvm-commits at lists.llvm.org
Thu Oct 1 01:48:57 PDT 2026
https://github.com/sommerlukas updated https://github.com/llvm/llvm-project/pull/227995
>From c552505d33361969a091f24f47797a2cb39e6e5d Mon Sep 17 00:00:00 2001
From: Jeffrey Byrnes <Jeffrey.Byrnes at amd.com>
Date: Wed, 8 Jul 2026 15:35:04 -0700
Subject: [PATCH 01/22] [AMDGPU] Prefetch up to 64kb ICache with
s_prefetch_inst_pcrel
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 34 +++++-
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h | 12 ++
llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp | 37 ++++++
.../AMDGPU/AsmParser/AMDGPUAsmParser.cpp | 23 ++++
.../AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp | 20 ++++
.../AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h | 8 ++
.../AMDGPU/MCTargetDesc/AMDGPUInstPrinter.cpp | 18 ++-
.../AMDGPU/MCTargetDesc/AMDGPUInstPrinter.h | 2 +
.../MCTargetDesc/AMDGPUMCCodeEmitter.cpp | 37 +++++-
.../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 90 ++++++++++++++
.../Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h | 21 ++++
llvm/lib/Target/AMDGPU/SIDefines.h | 3 +
.../lib/Target/AMDGPU/SIMachineFunctionInfo.h | 10 ++
llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp | 101 +++++++++++++++-
llvm/lib/Target/AMDGPU/SMInstructions.td | 5 +-
llvm/test/CodeGen/AMDGPU/icache-prefetch.ll | 112 ++++++++++++++++++
16 files changed, 518 insertions(+), 15 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index b24f62fa69a11..7ca9ebd5db622 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -193,6 +193,11 @@ void AMDGPUAsmPrinter::emitFunctionBodyStart() {
const GCNSubtarget &STM = MF->getSubtarget<GCNSubtarget>();
const Function &F = MF->getFunction();
+ // If ICache prefetch is enabled, create the function end symbol early so
+ // it can be referenced by the prefetch MCExprs during instruction emission.
+ if (MFI.hasICachePrefetch())
+ PrefetchEndSym = createTempSymbol("pref_func_end");
+
// TODO: We're checking this late, would be nice to check it earlier.
if (STM.requiresCodeObjectV6() && CodeObjectVersion < AMDGPU::AMDHSA_COV6) {
reportFatalUsageError(
@@ -220,6 +225,15 @@ void AMDGPUAsmPrinter::emitFunctionBodyStart() {
HSAMetadataStream->emitKernel(*MF, CurrentProgramInfo);
}
+void AMDGPUAsmPrinter::emitFunctionBodyEnd() {
+ // Emit the label for the prefetch end symbol if ICache prefetch is enabled.
+ // This symbol was created in emitFunctionBodyStart for this function.
+ if (PrefetchEndSym) {
+ OutStreamer->emitLabel(PrefetchEndSym);
+ PrefetchEndSym = nullptr;
+ }
+}
+
/// Set bits in a kernel descriptor MCExpr field:
/// return ((Dst & ~Mask) | (Value << Shift))
static const MCExpr *setBits(const MCExpr *Dst, const MCExpr *Value,
@@ -249,15 +263,23 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
// size. At this point .Lfunc_end has been emitted (by the base AsmPrinter)
// right after the function code, so (Lfunc_end - func_sym) gives the
// exact function code size in bytes.
+ //
+ // When ICache prefetch is enabled, we set INST_PREF_SIZE to 0 because
+ // the s_prefetch_inst instructions handle all prefetching.
if (STM.hasInstPrefSize()) {
- const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
- MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
- MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
-
uint32_t Mask, Shift, Width, CacheLineSize;
STM.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
- const MCExpr *InstPrefSize =
- AMDGPUMCExpr::createInstPrefSize(CodeSizeExpr, Ctx);
+
+ const MCExpr *InstPrefSize;
+ if (MFI.hasICachePrefetch()) {
+ // Disable hardware prefetch - s_prefetch_inst handles it.
+ InstPrefSize = MCConstantExpr::create(0, Ctx);
+ } else {
+ const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
+ MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
+ MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
+ InstPrefSize = AMDGPUMCExpr::createInstPrefSize(CodeSizeExpr, Ctx);
+ }
KD.compute_pgm_rsrc3 =
setBits(KD.compute_pgm_rsrc3, InstPrefSize, Mask, Shift, Ctx);
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
index 4394cde308665..871bd738b947b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
@@ -57,6 +57,10 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
MCCodeEmitter *DumpCodeInstEmitter = nullptr;
+ // Symbol for the function end, used by ICache prefetch MCExprs.
+ // Created early in emitFunctionBodyStart when prefetch is enabled.
+ MCSymbol *PrefetchEndSym = nullptr;
+
// When appropriate, add a _dvgpr$ symbol.
void emitDVgprSymbol(MachineFunction &MF);
@@ -139,6 +143,8 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
void emitFunctionBodyStart() override;
+ void emitFunctionBodyEnd() override;
+
void endFunction(const MachineFunction *MF);
void emitImplicitDef(const MachineInstr *MI) const override;
@@ -156,6 +162,12 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
bool PrintAsmOperand(const MachineInstr *MI, unsigned OpNo,
const char *ExtraCode, raw_ostream &O) override;
+ /// Get the symbol for the function end, used for ICache prefetch MCExprs.
+ MCSymbol *getPrefetchEndSym() const { return PrefetchEndSym; }
+
+ /// Get the code size estimate from SIProgramInfo.
+ uint64_t getCodeSize() { return CurrentProgramInfo.getFunctionCodeSize(*MF); }
+
protected:
void getAnalysisUsage(AnalysisUsage &AU) const override;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
index 1758db77d8a6e..fd4651a41f8a8 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
@@ -466,6 +466,43 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
MCInst TmpInst;
MCInstLowering.lower(MI, TmpInst);
+
+ // Fix up S_PREFETCH_INST_PC_REL instructions inserted by the ICache
+ // prefetch pass. Replace the slot index in the sdata operand with an
+ // MCExpr that computes the cacheline count based on exact code size.
+ if (MI->getOpcode() == AMDGPU::S_PREFETCH_INST_PC_REL) {
+ const SIMachineFunctionInfo *MFI = MF->getInfo<SIMachineFunctionInfo>();
+ if (MFI->hasICachePrefetch()) {
+ // Operand indices in the MCInst.
+ constexpr unsigned OffsetIdx = 0;
+ constexpr unsigned SdataIdx = 2;
+
+ // The sdata operand contains the slot index (0-15) set by the pass.
+ int64_t SlotIndex = TmpInst.getOperand(SdataIdx).getImm();
+
+ // Create MCExpr for code size using label subtraction.
+ // This gives the exact code size at assembly time.
+ const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
+ MCSymbolRefExpr::create(getPrefetchEndSym(), OutContext),
+ MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
+
+ // Create MCExpr for the slot index.
+ const MCExpr *SlotIndexExpr =
+ MCConstantExpr::create(SlotIndex, OutContext);
+
+ // Create MCExprs that will be evaluated at fixup time when symbol
+ // positions are known.
+ const MCExpr *CachelinesExpr = AMDGPUMCExpr::createPrefetchCachelines(
+ SlotIndexExpr, CodeSizeExpr, OutContext);
+ const MCExpr *OffsetExpr = AMDGPUMCExpr::createPrefetchOffset(
+ SlotIndexExpr, CodeSizeExpr, OutContext);
+
+ // Replace the offset and sdata operands with MCExprs.
+ TmpInst.getOperand(OffsetIdx) = MCOperand::createExpr(OffsetExpr);
+ TmpInst.getOperand(SdataIdx) = MCOperand::createExpr(CachelinesExpr);
+ }
+ }
+
EmitToStreamer(*OutStreamer, TmpInst);
if (DumpCodeInstEmitter) {
diff --git a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
index a01042e53bf37..9269eb15b0f5d 100644
--- a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
+++ b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
@@ -966,6 +966,7 @@ class AMDGPUOperand : public MCParsedAsmOperand {
bool isSMRDOffset8() const;
bool isSMEMOffset() const;
bool isSMRDLiteralOffset() const;
+ bool isPrefetchSdata() const;
bool isDPP8() const;
bool isDPPCtrl() const;
bool isBLGP() const;
@@ -9522,6 +9523,14 @@ bool AMDGPUOperand::isSMRDOffset8() const {
bool AMDGPUOperand::isSMEMOffset() const {
// Offset range is checked later by validator.
+ // Also accept prefetch offset expressions for ICache prefetch instructions.
+ if (isExpr()) {
+ if (Expr->getKind() == MCExpr::Target) {
+ const AMDGPUMCExpr *AExpr = static_cast<const AMDGPUMCExpr *>(Expr);
+ if (AExpr->getKind() == AMDGPUMCExpr::AGVK_PrefetchOffset)
+ return true;
+ }
+ }
return isImmLiteral();
}
@@ -9531,6 +9540,18 @@ bool AMDGPUOperand::isSMRDLiteralOffset() const {
return isImmLiteral() && !isUInt<8>(getImm()) && isUInt<32>(getImm());
}
+bool AMDGPUOperand::isPrefetchSdata() const {
+ // Accept immediates (u8) or prefetch cachelines expressions.
+ if (isExpr()) {
+ if (Expr->getKind() == MCExpr::Target) {
+ const AMDGPUMCExpr *AExpr = static_cast<const AMDGPUMCExpr *>(Expr);
+ if (AExpr->getKind() == AMDGPUMCExpr::AGVK_PrefetchCachelines)
+ return true;
+ }
+ }
+ return isImmLiteral() && isUInt<8>(getImm());
+}
+
//===----------------------------------------------------------------------===//
// vop3
//===----------------------------------------------------------------------===//
@@ -9611,6 +9632,8 @@ bool AMDGPUAsmParser::parsePrimaryExpr(const MCExpr *&Res, SMLoc &EndLoc) {
.Case("alignto", AGVK::AGVK_AlignTo)
.Case("occupancy", AGVK::AGVK_Occupancy)
.Case("instprefsize", AGVK::AGVK_InstPrefSize)
+ .Case("prefetchcachelines", AGVK::AGVK_PrefetchCachelines)
+ .Case("prefetchoffset", AGVK::AGVK_PrefetchOffset)
.Default(AGVK::AGVK_None);
if (VK != AGVK::AGVK_None && peekToken().is(AsmToken::LParen)) {
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp
index 5cd2aa86560ae..fbd15135da850 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp
@@ -91,6 +91,12 @@ static unsigned getFixupKindNumBytes(unsigned Kind) {
switch (Kind) {
case AMDGPU::fixup_si_sopp_br:
return 2;
+ case AMDGPU::fixup_si_prefetch_sdata:
+ // The sdata field is 5 bits at bit offset 6, spanning bytes 0 and 1.
+ return 2;
+ case AMDGPU::fixup_si_prefetch_offset:
+ // The offset field is 24 bits at bit offset 32 (bytes 4-6 of 8-byte inst).
+ return 3;
case FK_SecRel_1:
case FK_Data_1:
return 1;
@@ -121,6 +127,14 @@ static uint64_t adjustFixupValue(const MCFixup &Fixup, uint64_t Value,
return BrImm;
}
+ case AMDGPU::fixup_si_prefetch_sdata:
+ // The value is already the computed cacheline count from the MCExpr.
+ // Clamp to 5-bit field (max 31).
+ return std::min(Value, static_cast<uint64_t>(31));
+ case AMDGPU::fixup_si_prefetch_offset:
+ // The value is the byte offset. It's a 24-bit signed field.
+ // Clamp to valid range.
+ return SignedValue & 0xFFFFFF;
case FK_Data_1:
case FK_Data_2:
case FK_Data_4:
@@ -179,6 +193,12 @@ MCFixupKindInfo AMDGPUAsmBackend::getFixupKindInfo(MCFixupKind Kind) const {
const static MCFixupKindInfo Infos[AMDGPU::NumTargetFixupKinds] = {
// name offset bits flags
{"fixup_si_sopp_br", 0, 16, 0},
+ // Prefetch sdata is a 5-bit field at bits 10-6 in the 64-bit SMEM
+ // instruction encoding (GFX12+).
+ {"fixup_si_prefetch_sdata", 6, 5, 0},
+ // Prefetch offset is a 24-bit field at bit 0 of the high 32-bits
+ // (byte offset 4 from instruction start).
+ {"fixup_si_prefetch_offset", 0, 24, 0},
};
if (mc::isRelocation(Kind))
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h
index d49bb196ab3a1..ff98f69c7507e 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h
@@ -17,6 +17,14 @@ enum Fixups {
/// 16-bit PC relative fixup for SOPP branch instructions.
fixup_si_sopp_br = FirstTargetFixupKind,
+ /// Fixups for s_prefetch_inst instructions.
+ /// These compute values based on code size at assembly time.
+ /// The sdata field (cacheline count) is a 5-bit field.
+ fixup_si_prefetch_sdata,
+
+ /// The offset field (byte offset) is a 24-bit signed field.
+ fixup_si_prefetch_offset,
+
// Marker
LastTargetFixupKind,
NumTargetFixupKinds = LastTargetFixupKind - FirstTargetFixupKind
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.cpp
index 0aece53db1eb0..c9eed3f7158ee 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.cpp
@@ -145,7 +145,12 @@ void AMDGPUInstPrinter::printSMRDOffset8(const MCInst *MI, unsigned OpNo,
void AMDGPUInstPrinter::printSMEMOffset(const MCInst *MI, unsigned OpNo,
const MCSubtargetInfo &STI,
raw_ostream &O) {
- O << formatHex(MI->getOperand(OpNo).getImm());
+ const MCOperand &Op = MI->getOperand(OpNo);
+ if (Op.isExpr()) {
+ MAI.printExpr(O, *Op.getExpr());
+ } else {
+ O << formatHex(Op.getImm());
+ }
}
void AMDGPUInstPrinter::printSMRDLiteralOffset(const MCInst *MI, unsigned OpNo,
@@ -154,6 +159,17 @@ void AMDGPUInstPrinter::printSMRDLiteralOffset(const MCInst *MI, unsigned OpNo,
printU32ImmOperand(MI, OpNo, STI, O);
}
+void AMDGPUInstPrinter::printPrefetchSdata(const MCInst *MI, unsigned OpNo,
+ const MCSubtargetInfo &STI,
+ raw_ostream &O) {
+ const MCOperand &Op = MI->getOperand(OpNo);
+ if (Op.isExpr()) {
+ MAI.printExpr(O, *Op.getExpr());
+ } else {
+ O << Op.getImm();
+ }
+}
+
void AMDGPUInstPrinter::printCPol(const MCInst *MI, unsigned OpNo,
const MCSubtargetInfo &STI, raw_ostream &O) {
auto Imm = MI->getOperand(OpNo).getImm();
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.h
index 5f5f15712b5ac..599070c70b428 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.h
@@ -59,6 +59,8 @@ class AMDGPUInstPrinter : public MCInstPrinter {
const MCSubtargetInfo &STI, raw_ostream &O);
void printSMRDLiteralOffset(const MCInst *MI, unsigned OpNo,
const MCSubtargetInfo &STI, raw_ostream &O);
+ void printPrefetchSdata(const MCInst *MI, unsigned OpNo,
+ const MCSubtargetInfo &STI, raw_ostream &O);
void printCPol(const MCInst *MI, unsigned OpNo,
const MCSubtargetInfo &STI, raw_ostream &O);
void printTH(const MCInst *MI, int64_t TH, int64_t Scope, raw_ostream &O);
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCCodeEmitter.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCCodeEmitter.cpp
index 18c0919471d6d..328c02b285d60 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCCodeEmitter.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCCodeEmitter.cpp
@@ -513,7 +513,19 @@ void AMDGPUMCCodeEmitter::getSOPPBrEncoding(const MCInst &MI, unsigned OpNo,
void AMDGPUMCCodeEmitter::getSMEMOffsetEncoding(
const MCInst &MI, unsigned OpNo, APInt &Op,
SmallVectorImpl<MCFixup> &Fixups, const MCSubtargetInfo &STI) const {
- auto Offset = MI.getOperand(OpNo).getImm();
+ const MCOperand &MO = MI.getOperand(OpNo);
+ if (MO.isExpr()) {
+ const MCExpr *Expr = MO.getExpr();
+ if (Expr->getKind() == MCExpr::Target) {
+ const AMDGPUMCExpr *AExpr = static_cast<const AMDGPUMCExpr *>(Expr);
+ if (AExpr->getKind() == AMDGPUMCExpr::AGVK_PrefetchOffset) {
+ addFixup(Fixups, 4, Expr, AMDGPU::fixup_si_prefetch_offset);
+ Op = 0;
+ return;
+ }
+ }
+ }
+ auto Offset = MO.getImm();
// VI only supports 20-bit unsigned offsets.
assert(!AMDGPU::isVI(STI) || isUInt<20>(Offset));
Op = Offset;
@@ -716,6 +728,29 @@ void AMDGPUMCCodeEmitter::getMachineOpValueCommon(
} else if (MO.isExpr() && MO.getExpr()->evaluateAsAbsolute(Val)) {
isLikeImm = true;
} else if (MO.isExpr()) {
+ // Check for prefetch cacheline MCExpr - these need a special fixup
+ // because the expression computes a cacheline count based on code size.
+ const MCExpr *Expr = MO.getExpr();
+ if (Expr->getKind() == MCExpr::Target) {
+ const AMDGPUMCExpr *AExpr = static_cast<const AMDGPUMCExpr *>(Expr);
+ if (AExpr->getKind() == AMDGPUMCExpr::AGVK_PrefetchCachelines) {
+ // Create a fixup for the prefetch sdata field. The fixup will be
+ // applied after assembly when symbol positions are known.
+ // Offset 0 means from the start of the instruction.
+ addFixup(Fixups, 0, Expr, AMDGPU::fixup_si_prefetch_sdata);
+ // Set Op to 0 as placeholder; fixup will overwrite.
+ Op = 0;
+ return;
+ }
+ if (AExpr->getKind() == AMDGPUMCExpr::AGVK_PrefetchOffset) {
+ // Create a fixup for the prefetch offset field.
+ // Offset 4 means byte 4 of the 8-byte instruction.
+ addFixup(Fixups, 4, Expr, AMDGPU::fixup_si_prefetch_offset);
+ Op = 0;
+ return;
+ }
+ }
+
// FIXME: If this is expression is PCRel or not should not depend on what
// the expression looks like. Given that this is just a general expression,
// it should probably be FK_Data_4 and whatever is producing
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index 745647e235f6e..17028fc9b26d9 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -68,6 +68,8 @@ unsigned AMDGPUMCExpr::getNumExpectedArgs(VariantKind Kind) {
return 1;
case AGVK_TotalNumVGPRs:
case AGVK_AlignTo:
+ case AGVK_PrefetchCachelines:
+ case AGVK_PrefetchOffset:
return 2;
case AGVK_ExtraSGPRs:
return 3;
@@ -110,6 +112,12 @@ void AMDGPUMCExpr::printImpl(raw_ostream &OS, const MCAsmInfo *MAI) const {
case AGVK_InstPrefSize:
OS << "instprefsize(";
break;
+ case AGVK_PrefetchCachelines:
+ OS << "prefetchcachelines(";
+ break;
+ case AGVK_PrefetchOffset:
+ OS << "prefetchoffset(";
+ break;
case AGVK_Lit:
OS << "lit(";
break;
@@ -245,6 +253,74 @@ bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
return true;
}
+bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
+ const MCAssembler *Asm) const {
+ uint64_t SlotIndex = 0, CodeSizeInBytes = 0;
+ if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes}))
+ return false;
+
+ // Constants for prefetch calculation.
+ // Each instruction can prefetch up to 31 cachelines (5-bit sdata field).
+ // Cacheline size is 128 bytes. Each slot covers ~4KB (31 * 128 = 3968 bytes).
+ constexpr unsigned MaxCachelinesPerPrefetch = 31;
+ constexpr unsigned CacheLineSize = 128;
+ constexpr unsigned BytesPerPrefetch =
+ MaxCachelinesPerPrefetch * CacheLineSize;
+ constexpr uint64_t MaxPrefetchSize = 64 * 1024; // 64KB ICache
+
+ // Clamp code size to maximum prefetchable size.
+ uint64_t PrefetchSize = std::min(CodeSizeInBytes, MaxPrefetchSize);
+
+ // Calculate the byte offset for this slot.
+ uint64_t SlotOffset = SlotIndex * BytesPerPrefetch;
+
+ // If this slot starts beyond the code size, return 0 (NOP).
+ if (SlotOffset >= PrefetchSize) {
+ Res = MCValue::get(static_cast<int64_t>(0));
+ return true;
+ }
+
+ // Calculate remaining bytes from this slot's offset.
+ uint64_t RemainingBytes = PrefetchSize - SlotOffset;
+
+ // Calculate cachelines needed, clamped to max per instruction.
+ uint64_t CachelinesNeeded = divideCeil(RemainingBytes, CacheLineSize);
+ uint64_t CachelineCount = std::min(
+ CachelinesNeeded, static_cast<uint64_t>(MaxCachelinesPerPrefetch));
+
+ Res = MCValue::get(static_cast<int64_t>(CachelineCount));
+ return true;
+}
+
+bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
+ const MCAssembler *Asm) const {
+ uint64_t SlotIndex = 0, CodeSizeInBytes = 0;
+ if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes}))
+ return false;
+
+ constexpr unsigned MaxCachelinesPerPrefetch = 31;
+ constexpr unsigned CacheLineSize = 128;
+ constexpr uint64_t MaxPrefetchSize = 64 * 1024;
+
+ uint64_t PrefetchSize = std::min(CodeSizeInBytes, MaxPrefetchSize);
+ uint64_t Offset = 0;
+ uint64_t Remaining = PrefetchSize;
+
+ for (uint64_t I = 0; I < SlotIndex; ++I) {
+ if (Remaining == 0)
+ break;
+ uint64_t Cachelines =
+ std::min(divideCeil(Remaining, CacheLineSize),
+ static_cast<uint64_t>(MaxCachelinesPerPrefetch));
+ uint64_t PrefetchBytes = Cachelines * CacheLineSize;
+ Offset += PrefetchBytes;
+ Remaining = (Remaining > PrefetchBytes) ? Remaining - PrefetchBytes : 0;
+ }
+
+ Res = MCValue::get(static_cast<int64_t>(Offset));
+ return true;
+}
+
bool AMDGPUMCExpr::isSymbolUsedInExpression(const MCSymbol *Sym,
const MCExpr *E) {
switch (E->getKind()) {
@@ -292,6 +368,10 @@ bool AMDGPUMCExpr::evaluateAsRelocatableImpl(MCValue &Res,
return evaluateOccupancy(Res, Asm);
case AGVK_InstPrefSize:
return evaluateInstPrefSize(Res, Asm);
+ case AGVK_PrefetchCachelines:
+ return evaluatePrefetchCachelines(Res, Asm);
+ case AGVK_PrefetchOffset:
+ return evaluatePrefetchOffset(Res, Asm);
case AGVK_Lit:
case AGVK_Lit64:
return Args[0]->evaluateAsRelocatable(Res, Asm);
@@ -349,6 +429,16 @@ AMDGPUMCExpr::createInstPrefSize(const MCExpr *CodeSizeBytes, MCContext &Ctx) {
return create(AGVK_InstPrefSize, {CodeSizeBytes}, Ctx);
}
+const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchCachelines(
+ const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes, MCContext &Ctx) {
+ return create(AGVK_PrefetchCachelines, {SlotIndex, CodeSizeBytes}, Ctx);
+}
+
+const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchOffset(
+ const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes, MCContext &Ctx) {
+ return create(AGVK_PrefetchOffset, {SlotIndex, CodeSizeBytes}, Ctx);
+}
+
const AMDGPUMCExpr *AMDGPUMCExpr::createLit(LitModifier Lit, int64_t Value,
MCContext &Ctx) {
assert(Lit == LitModifier::Lit || Lit == LitModifier::Lit64);
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index 96003711faf88..a8d9385121b03 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -41,6 +41,8 @@ class AMDGPUMCExpr : public MCTargetExpr {
AGVK_AlignTo,
AGVK_Occupancy,
AGVK_InstPrefSize,
+ AGVK_PrefetchCachelines,
+ AGVK_PrefetchOffset,
AGVK_Lit,
AGVK_Lit64,
AGVK_Min,
@@ -74,6 +76,8 @@ class AMDGPUMCExpr : public MCTargetExpr {
bool evaluateAlignTo(MCValue &Res, const MCAssembler *Asm) const;
bool evaluateOccupancy(MCValue &Res, const MCAssembler *Asm) const;
bool evaluateInstPrefSize(MCValue &Res, const MCAssembler *Asm) const;
+ bool evaluatePrefetchCachelines(MCValue &Res, const MCAssembler *Asm) const;
+ bool evaluatePrefetchOffset(MCValue &Res, const MCAssembler *Asm) const;
public:
static const AMDGPUMCExpr *
@@ -116,6 +120,23 @@ class AMDGPUMCExpr : public MCTargetExpr {
static const AMDGPUMCExpr *createInstPrefSize(const MCExpr *CodeSizeBytes,
MCContext &Ctx);
+ /// Create an expression for computing cacheline count for a prefetch slot.
+ /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
+ /// CodeSizeBytes is the total code size in bytes.
+ /// Returns the number of cachelines this slot should prefetch.
+ static const AMDGPUMCExpr *
+ createPrefetchCachelines(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
+ MCContext &Ctx);
+
+ /// Create an expression for computing the byte offset for a prefetch slot.
+ /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
+ /// CodeSizeBytes is the total code size in bytes.
+ /// Returns the cumulative byte offset where this slot should start
+ /// prefetching.
+ static const AMDGPUMCExpr *createPrefetchOffset(const MCExpr *SlotIndex,
+ const MCExpr *CodeSizeBytes,
+ MCContext &Ctx);
+
static const AMDGPUMCExpr *createLit(LitModifier Lit, int64_t Value,
MCContext &Ctx);
diff --git a/llvm/lib/Target/AMDGPU/SIDefines.h b/llvm/lib/Target/AMDGPU/SIDefines.h
index 864a84cc8daf4..a67c259541378 100644
--- a/llvm/lib/Target/AMDGPU/SIDefines.h
+++ b/llvm/lib/Target/AMDGPU/SIDefines.h
@@ -824,6 +824,9 @@ enum ModeRegisterMasks : uint32_t {
SRC2_VGPR_MSB = 0x3 << 18,
VGPR_MSB_MASK = 0xff << 12, // Bits 12..19
+ // GFX1250 instruction prefetch
+ SCALAR_PREFETCH_EN = 1 << 24,
+
REPLAY_MODE = 1 << 25,
FLAT_SCRATCH_IS_NV = 1 << 26,
};
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
index 0207c728ea9b8..4b85168bdfed6 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
@@ -490,6 +490,10 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
bool HasNonSpillStackObjects = false;
bool IsStackRealigned = false;
+ // Set when ICache prefetch instructions have been inserted in the entry
+ // block. This tells AsmPrinter to set rsrc3 INST_PREF_SIZE to 0.
+ bool HasICachePrefetch = false;
+
unsigned NumSpilledSGPRs = 0;
unsigned NumSpilledVGPRs = 0;
@@ -1126,6 +1130,12 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
IsStackRealigned = Realigned;
}
+ bool hasICachePrefetch() const { return HasICachePrefetch; }
+
+ void setHasICachePrefetch(bool Prefetch = true) {
+ HasICachePrefetch = Prefetch;
+ }
+
unsigned getNumSpilledSGPRs() const {
return NumSpilledSGPRs;
}
diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index 9b67cdd6be6f4..608f03b1dedce 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -20,12 +20,17 @@
#include "AMDGPU.h"
#include "GCNSubtarget.h"
+#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
+#include "SIMachineFunctionInfo.h"
#include "llvm/ADT/Statistic.h"
#include "llvm/CodeGen/MachineDominators.h"
#include "llvm/CodeGen/MachineFunctionPass.h"
+#include "llvm/CodeGen/MachineInstrBuilder.h"
#include "llvm/CodeGen/MachineLoopInfo.h"
#include "llvm/CodeGen/TargetSchedule.h"
#include "llvm/Support/BranchProbability.h"
+#include "llvm/Support/CommandLine.h"
+#include "llvm/TargetParser/Triple.h"
using namespace llvm;
#define DEBUG_TYPE "si-pre-emit-peephole"
@@ -45,12 +50,26 @@ struct ModeFieldState {
bool isTracked() const { return PendingWrite || Value; }
};
+static cl::opt<bool>
+ EnableICachePrefetch("amdgpu-icache-prefetch",
+ cl::desc("Insert ICache prefetch instructions"),
+ cl::init(true), cl::Hidden);
+
+namespace {
+
+// Number of prefetch instructions to insert.
+// Each can prefetch up to 31 cachelines of 128 bytes = ~4KB.
+// 16 instructions cover 64KB (the full ICache size).
+static constexpr unsigned NumPrefetchInsts = 16;
+
class SIPreEmitPeephole {
private:
+ const GCNSubtarget *ST = nullptr;
const SIInstrInfo *TII = nullptr;
const SIRegisterInfo *TRI = nullptr;
MachineLoopInfo *MLI = nullptr;
+ bool insertICachePrefetch(MachineFunction &MF);
bool optimizeVccBranch(MachineInstr &MI) const;
void updateMLIBeforeRemovingEdge(MachineBasicBlock *From,
MachineBasicBlock *To) const;
@@ -191,8 +210,7 @@ bool SIPreEmitPeephole::optimizeVccBranch(MachineInstr &MI) const {
bool Changed = false;
MachineBasicBlock &MBB = *MI.getParent();
- const GCNSubtarget &ST = MBB.getParent()->getSubtarget<GCNSubtarget>();
- const bool IsWave32 = ST.isWave32();
+ const bool IsWave32 = ST->isWave32();
const unsigned CondReg = TRI->getVCC();
const unsigned ExecReg = IsWave32 ? AMDGPU::EXEC_LO : AMDGPU::EXEC;
const unsigned And = IsWave32 ? AMDGPU::S_AND_B32 : AMDGPU::S_AND_B64;
@@ -850,6 +868,73 @@ MachineInstrBuilder SIPreEmitPeephole::createUnpackedMI(MachineInstr &I,
return NewMI;
}
+bool SIPreEmitPeephole::insertICachePrefetch(MachineFunction &MF) {
+ // Only run on targets that support ICache prefetching.
+ if (!ST->hasICachePrefetch())
+ return false;
+
+ // Only run for AMDHSA - this is where kernel descriptors are used
+ // and rsrc3 INST_PREF_SIZE is relevant.
+ const Triple &TT = ST->getTargetTriple();
+ if (TT.getOS() != Triple::AMDHSA)
+ return false;
+
+ SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
+
+ // Only insert prefetch instructions for entry functions.
+ if (!MFI->isEntryFunction())
+ return false;
+
+ MachineBasicBlock &EntryBB = MF.front();
+ MachineBasicBlock::iterator InsertPt = EntryBB.begin();
+
+ // Skip past any instructions that must remain at the very beginning:
+ // - Debug values and CFI instructions
+ // - S_SETREG_IMM32_B32 instructions that set up MODE register bits
+ // (e.g., REPLAY_MODE bit 25 from SIFrameLowering)
+ // We want the prefetches to come after all initial MODE setup.
+ while (InsertPt != EntryBB.end()) {
+ if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction()) {
+ ++InsertPt;
+ continue;
+ }
+ if (InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32) {
+ ++InsertPt;
+ continue;
+ }
+ break;
+ }
+
+ DebugLoc DL;
+
+ // Insert s_setreg_imm32_b32 to set MODE.SCALAR_PREFETCH_EN (bit 24).
+ // This ensures only the first wave in the WGP executes the prefetches.
+ // Insert it right before the prefetch instructions (after other MODE setup).
+ using namespace AMDGPU::Hwreg;
+ unsigned ModeRegEncoding = HwregEncoding::encode(ID_MODE, 24, 1);
+ BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_SETREG_IMM32_B32))
+ .addImm(1) // Value to set (enable)
+ .addImm(ModeRegEncoding);
+
+ // Insert 16 s_prefetch_inst_pc_rel instructions.
+ // The offset and sdata operands are placeholders - the sdata operand stores
+ // the slot index (0-15). Both will be fixed up in AMDGPUAsmPrinter based on
+ // the actual code size.
+ for (unsigned I = 0; I < NumPrefetchInsts; ++I) {
+ BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
+ .addImm(0) // offset (placeholder, fixed up later)
+ .addReg(AMDGPU::SGPR_NULL) // soffset
+ .addImm(I); // sdata (slot index, fixed up later)
+ }
+
+ // Mark that we've inserted ICache prefetch instructions.
+ // This tells AsmPrinter to set rsrc3 INST_PREF_SIZE to 0 and fix up
+ // the prefetch cacheline counts.
+ MFI->setHasICachePrefetch(true);
+
+ return true;
+}
+
PreservedAnalyses
llvm::SIPreEmitPeepholePass::run(MachineFunction &MF,
MachineFunctionAnalysisManager &MFAM) {
@@ -866,12 +951,16 @@ llvm::SIPreEmitPeepholePass::run(MachineFunction &MF,
}
bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
- const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
- TII = ST.getInstrInfo();
+ ST = &MF.getSubtarget<GCNSubtarget>();
+ TII = ST->getInstrInfo();
TRI = &TII->getRegisterInfo();
MLI = LoopInfo;
bool Changed = false;
+ // Insert ICache prefetch instructions if enabled.
+ if (EnableICachePrefetch)
+ Changed |= insertICachePrefetch(MF);
+
MF.RenumberBlocks();
for (MachineBasicBlock &MBB : MF) {
@@ -892,7 +981,7 @@ bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
}
}
- if (!ST.hasVGPRIndexMode())
+ if (!ST->hasVGPRIndexMode())
continue;
MachineInstr *SetGPRMI = nullptr;
@@ -929,7 +1018,7 @@ bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
// side effects.
// Perform the extra MF scans only for supported archs
- if (!ST.hasGFX940Insts())
+ if (!ST->hasGFX940Insts())
return Changed;
for (MachineBasicBlock &MBB : MF) {
// Unpack packed instructions overlapped by MFMAs. This allows the
diff --git a/llvm/lib/Target/AMDGPU/SMInstructions.td b/llvm/lib/Target/AMDGPU/SMInstructions.td
index 19aeafe9b30cc..7f75c77b9ea0c 100644
--- a/llvm/lib/Target/AMDGPU/SMInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SMInstructions.td
@@ -8,6 +8,9 @@
def smrd_offset_8 : ImmOperand<i32, "SMRDOffset8", 1>;
+// Prefetch sdata operand - accepts immediates or MCExpr for ICache prefetch.
+def PrefetchSdata : ImmOperand<i8, "PrefetchSdata", 1>;
+
let EncoderMethod = "getSMEMOffsetEncoding",
DecoderMethod = "decodeSMEMOffset" in {
def SMEMOffset : ImmOperand<i32, "SMEMOffset", 1>;
@@ -239,7 +242,7 @@ class SM_WaveId_Pseudo<string opName, SDPatternOperator node> : SM_Pseudo<
class SM_Prefetch_Pseudo <string opName, RegisterClass baseClass, bit hasSBase>
: SM_Pseudo<opName, (outs), !con(!if(hasSBase, (ins baseClass:$sbase), (ins)),
- (ins SMEMOffset:$offset, SReg_32:$soffset, i8imm:$sdata, CPol_0:$cpol)),
+ (ins SMEMOffset:$offset, SReg_32:$soffset, PrefetchSdata:$sdata, CPol_0:$cpol)),
!if(hasSBase, " $sbase,", "") # " $offset, $soffset, $sdata$cpol"> {
// Mark prefetches as both load and store to prevent reordering with loads
// and stores. This is also needed for pattern to match prefetch intrinsic.
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
new file mode 100644
index 0000000000000..2dc738cd703a6
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -0,0 +1,112 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -o - %s | \
+; RUN: FileCheck -check-prefix=GFX1250 %s
+
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=false -o - %s | \
+; RUN: FileCheck -check-prefix=NO-PREFETCH %s
+
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
+; RUN: llvm-objdump -d %t.o | FileCheck -check-prefix=GFX1250-OBJ %s
+
+; Test that ICache prefetch instructions are inserted for entry functions on gfx1250.
+; Verify object file has resolved cacheline counts.
+; GFX1250-OBJ-LABEL: <entry_kernel>:
+; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x0, null, 2
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x100, null, 0
+
+define amdgpu_kernel void @entry_kernel(ptr addrspace(1) %out) {
+; GFX1250-LABEL: entry_kernel:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 24, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(0, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(1, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(2, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(3, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(4, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(5, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(6, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(7, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(8, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(9, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(10, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(11, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(12, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(13, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(14, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(15, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
+; GFX1250-NEXT: v_mov_b32_e32 v0, 0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: global_store_b32 v0, v0, s[0:1]
+; GFX1250-NEXT: s_endpgm
+; GFX1250-NEXT: .Lpref_func_end0:
+;
+; NO-PREFETCH-LABEL: entry_kernel:
+; NO-PREFETCH: ; %bb.0:
+; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
+; NO-PREFETCH-NEXT: v_mov_b32_e32 v0, 0
+; NO-PREFETCH-NEXT: s_wait_kmcnt 0x0
+; NO-PREFETCH-NEXT: global_store_b32 v0, v0, s[0:1]
+; NO-PREFETCH-NEXT: s_endpgm
+ store i32 0, ptr addrspace(1) %out
+ ret void
+}
+
+; Verify that called functions do NOT get prefetch instructions.
+
+define void @called_function(ptr addrspace(1) %out) {
+; GFX1250-LABEL: called_function:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: v_mov_b32_e32 v2, 1
+; GFX1250-NEXT: global_store_b32 v[0:1], v2, off
+; GFX1250-NEXT: s_set_pc_i64 s[30:31]
+;
+; NO-PREFETCH-LABEL: called_function:
+; NO-PREFETCH: ; %bb.0:
+; NO-PREFETCH-NEXT: s_wait_loadcnt_dscnt 0x0
+; NO-PREFETCH-NEXT: s_wait_kmcnt 0x0
+; NO-PREFETCH-NEXT: v_mov_b32_e32 v2, 1
+; NO-PREFETCH-NEXT: global_store_b32 v[0:1], v2, off
+; NO-PREFETCH-NEXT: s_set_pc_i64 s[30:31]
+ store i32 1, ptr addrspace(1) %out
+ ret void
+}
+
+; GFX1250-OBJ-LABEL: <tiny_kernel>:
+; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x0, null, 2
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x100, null, 0
+
+define amdgpu_kernel void @tiny_kernel() {
+; GFX1250-LABEL: tiny_kernel:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 24, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(0, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(1, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(2, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(3, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(4, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(5, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(6, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(7, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(8, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(9, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(10, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(11, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(12, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(13, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(14, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(15, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT: s_endpgm
+; GFX1250-NEXT: .Lpref_func_end1:
+;
+; NO-PREFETCH-LABEL: tiny_kernel:
+; NO-PREFETCH: ; %bb.0:
+; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT: s_endpgm
+ ret void
+}
>From 1d4a3033cc0b28997b9f8f51c8c2b3968ac71dc8 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 30 Jul 2026 11:35:12 -0500
Subject: [PATCH 02/22] Account for sdata increment
Instruction semantics for `s_prefetch_inst` adds 1 to the encoded
`sdata` for the number of cache lines to prefetch. Account for that
increment in the calculations.
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
.../Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp | 4 ++--
.../lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h | 2 +-
llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 11 ++++++-----
llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h | 6 ++++--
llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp | 6 +++---
5 files changed, 16 insertions(+), 13 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp
index fbd15135da850..1acf18b757e0f 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp
@@ -128,8 +128,8 @@ static uint64_t adjustFixupValue(const MCFixup &Fixup, uint64_t Value,
return BrImm;
}
case AMDGPU::fixup_si_prefetch_sdata:
- // The value is already the computed cacheline count from the MCExpr.
- // Clamp to 5-bit field (max 31).
+ // The value is already the encoded sdata field value from the MCExpr.
+ // Clamp to the maximum 5-bit encoded field value (31).
return std::min(Value, static_cast<uint64_t>(31));
case AMDGPU::fixup_si_prefetch_offset:
// The value is the byte offset. It's a 24-bit signed field.
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h
index ff98f69c7507e..caa3930411a9b 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h
@@ -19,7 +19,7 @@ enum Fixups {
/// Fixups for s_prefetch_inst instructions.
/// These compute values based on code size at assembly time.
- /// The sdata field (cacheline count) is a 5-bit field.
+ /// The encoded 5-bit sdata field (cacheline count minus one).
fixup_si_prefetch_sdata,
/// The offset field (byte offset) is a 24-bit signed field.
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index 17028fc9b26d9..9459f3e32ac1c 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -260,9 +260,9 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
return false;
// Constants for prefetch calculation.
- // Each instruction can prefetch up to 31 cachelines (5-bit sdata field).
- // Cacheline size is 128 bytes. Each slot covers ~4KB (31 * 128 = 3968 bytes).
- constexpr unsigned MaxCachelinesPerPrefetch = 31;
+ // Each instruction can prefetch up to 32 cachelines (5-bit sdata field, plus
+ // one added). Cacheline size is 128 bytes. Each slot covers 4KiB.
+ constexpr unsigned MaxCachelinesPerPrefetch = 32;
constexpr unsigned CacheLineSize = 128;
constexpr unsigned BytesPerPrefetch =
MaxCachelinesPerPrefetch * CacheLineSize;
@@ -288,7 +288,8 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
uint64_t CachelineCount = std::min(
CachelinesNeeded, static_cast<uint64_t>(MaxCachelinesPerPrefetch));
- Res = MCValue::get(static_cast<int64_t>(CachelineCount));
+ // The instruction adds 1 to the encoded sdata, so deduct it here.
+ Res = MCValue::get(static_cast<int64_t>(CachelineCount - 1));
return true;
}
@@ -298,7 +299,7 @@ bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes}))
return false;
- constexpr unsigned MaxCachelinesPerPrefetch = 31;
+ constexpr unsigned MaxCachelinesPerPrefetch = 32;
constexpr unsigned CacheLineSize = 128;
constexpr uint64_t MaxPrefetchSize = 64 * 1024;
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index a8d9385121b03..d3693fb14045e 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -120,10 +120,12 @@ class AMDGPUMCExpr : public MCTargetExpr {
static const AMDGPUMCExpr *createInstPrefSize(const MCExpr *CodeSizeBytes,
MCContext &Ctx);
- /// Create an expression for computing cacheline count for a prefetch slot.
+ /// Create an expression for computing the encoded sdata field for a prefetch
+ /// slot.
/// SlotIndex is the 0-based index of the prefetch instruction (0-15).
/// CodeSizeBytes is the total code size in bytes.
- /// Returns the number of cachelines this slot should prefetch.
+ /// Returns the requested cacheline count minus one, encoded for the 5-bit
+ /// sdata field.
static const AMDGPUMCExpr *
createPrefetchCachelines(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
MCContext &Ctx);
diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index 608f03b1dedce..23745b575158e 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -58,9 +58,9 @@ static cl::opt<bool>
namespace {
// Number of prefetch instructions to insert.
-// Each can prefetch up to 31 cachelines of 128 bytes = ~4KB.
-// 16 instructions cover 64KB (the full ICache size).
-static constexpr unsigned NumPrefetchInsts = 16;
+// Each can prefetch up to 32 cachelines of 128 bytes = 4KiB.
+// 16 instructions cover 64KiB (the full ICache size).
+static constexpr unsigned MaxNumPrefetchInsts = 16;
class SIPreEmitPeephole {
private:
>From aa014988acaf08c501fa18e8b672a408407cac31 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 30 Jul 2026 11:41:09 -0500
Subject: [PATCH 03/22] Use code size estimate for prefetch insertion
When inserting explicit instruction cache prefetch instructions, take
the estimated code size into account:
- If the program is less than 32KiB, use the previous mechanism in the
kernel descriptor instead of explicit instructions.
- Limit the number of prefetch instructions inserted to the necessary
number plus some slack.
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp | 47 ++++++++++++--------
1 file changed, 28 insertions(+), 19 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index 23745b575158e..b5bbb6dd23f4b 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -22,6 +22,7 @@
#include "GCNSubtarget.h"
#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
#include "SIMachineFunctionInfo.h"
+#include "SIProgramInfo.h"
#include "llvm/ADT/Statistic.h"
#include "llvm/CodeGen/MachineDominators.h"
#include "llvm/CodeGen/MachineFunctionPass.h"
@@ -57,11 +58,6 @@ static cl::opt<bool>
namespace {
-// Number of prefetch instructions to insert.
-// Each can prefetch up to 32 cachelines of 128 bytes = 4KiB.
-// 16 instructions cover 64KiB (the full ICache size).
-static constexpr unsigned MaxNumPrefetchInsts = 16;
-
class SIPreEmitPeephole {
private:
const GCNSubtarget *ST = nullptr;
@@ -885,6 +881,16 @@ bool SIPreEmitPeephole::insertICachePrefetch(MachineFunction &MF) {
if (!MFI->isEntryFunction())
return false;
+ SIProgramInfo PI;
+ uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
+ // The kernel descriptor can specify an instruction prefetch size of up to 256
+ // in INST_PREF_SIZE. At a granularity of 128B, this equals 32KiB of
+ // instructions that can be prefetched without inserting explicit prefetch
+ // instructions.
+ constexpr uint64_t MaxKDPrefetch = 1u << 15;
+ if (ProgramSize <= MaxKDPrefetch)
+ return false;
+
MachineBasicBlock &EntryBB = MF.front();
MachineBasicBlock::iterator InsertPt = EntryBB.begin();
@@ -907,20 +913,23 @@ bool SIPreEmitPeephole::insertICachePrefetch(MachineFunction &MF) {
DebugLoc DL;
- // Insert s_setreg_imm32_b32 to set MODE.SCALAR_PREFETCH_EN (bit 24).
- // This ensures only the first wave in the WGP executes the prefetches.
- // Insert it right before the prefetch instructions (after other MODE setup).
- using namespace AMDGPU::Hwreg;
- unsigned ModeRegEncoding = HwregEncoding::encode(ID_MODE, 24, 1);
- BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_SETREG_IMM32_B32))
- .addImm(1) // Value to set (enable)
- .addImm(ModeRegEncoding);
-
- // Insert 16 s_prefetch_inst_pc_rel instructions.
- // The offset and sdata operands are placeholders - the sdata operand stores
- // the slot index (0-15). Both will be fixed up in AMDGPUAsmPrinter based on
- // the actual code size.
- for (unsigned I = 0; I < NumPrefetchInsts; ++I) {
+ // Calculate the number of prefetch instructions required for the current
+ // program size. Each prefetch can transfer 4KiB of instructions. Add
+ // some slack as inserting the prefetches and later transformations, e.g.,
+ // padding and alignment, will introduce additional bytes. The offset and
+ // sdata operands are placeholders - the sdata operand stores the slot index
+ // (0-15). Both will be fixed up in AMDGPUAsmPrinter based on the actual code
+ // size.
+ constexpr uint64_t PrefetchSlack = 2 * 1024;
+ constexpr uint64_t BytesPerPrefetch = 4 * 1024;
+ // Each prefetch can transfer up to 32 cachelines of 128 bytes = 4KiB.
+ // 16 instructions cover 64KiB (the full ICache size).
+ constexpr unsigned MaxNumPrefetchInsts = 16;
+ unsigned NumPrefetches =
+ llvm::divideCeil(ProgramSize + PrefetchSlack, BytesPerPrefetch);
+ // Limit to 16 prefetches at most, otherwise we'd exceed the cache size.
+ NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
+ for (unsigned I = 0; I < NumPrefetches; ++I) {
BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
.addImm(0) // offset (placeholder, fixed up later)
.addReg(AMDGPU::SGPR_NULL) // soffset
>From 74db290de8208f0d8c45276454c90143031eba3a Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 31 Jul 2026 03:33:29 -0500
Subject: [PATCH 04/22] Set INST_PREF_SIZE to 1 for explicit prefetch
Setting INST_PREF_SIZE to 1 is safe and activates the prefetching
feature. This means the first wave on the WGP will have its
SCALAR_PREFETCH_EN set to 1, while it is deactivated for the other
waves.
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 10 ++++++----
llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp | 2 +-
2 files changed, 7 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 7ca9ebd5db622..e3bf41f8e513f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -264,16 +264,18 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
// right after the function code, so (Lfunc_end - func_sym) gives the
// exact function code size in bytes.
//
- // When ICache prefetch is enabled, we set INST_PREF_SIZE to 0 because
- // the s_prefetch_inst instructions handle all prefetching.
+ // When ICache prefetch is enabled, we set INST_PREF_SIZE to 1 because
+ // the s_prefetch_inst instructions handle the actual prefetching.
if (STM.hasInstPrefSize()) {
uint32_t Mask, Shift, Width, CacheLineSize;
STM.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
const MCExpr *InstPrefSize;
if (MFI.hasICachePrefetch()) {
- // Disable hardware prefetch - s_prefetch_inst handles it.
- InstPrefSize = MCConstantExpr::create(0, Ctx);
+ // Set INST_PREF_SIZE to 1. This enables prefetch and the first wave on
+ // the WGP gets SCALAR_PREFETCH_EN set to 1, while the remaining waves
+ // receive 0. This enables the s_prefetch_inst for the first wave.
+ InstPrefSize = MCConstantExpr::create(1, Ctx);
} else {
const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index b5bbb6dd23f4b..11344ff8e4709 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -937,7 +937,7 @@ bool SIPreEmitPeephole::insertICachePrefetch(MachineFunction &MF) {
}
// Mark that we've inserted ICache prefetch instructions.
- // This tells AsmPrinter to set rsrc3 INST_PREF_SIZE to 0 and fix up
+ // This tells AsmPrinter to set rsrc3 INST_PREF_SIZE to 1 and fix up
// the prefetch cacheline counts.
MFI->setHasICachePrefetch(true);
>From 0cb3b7013ec61921f9a8feb5d1c768999da82bc6 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 31 Jul 2026 08:24:19 -0500
Subject: [PATCH 05/22] Account for PC in offset calculcation
The `s_prefetch_inst_pc_rel` instruction prefetches relative to its own
PC. Account for that in the offset calculation to avoid gaps in the
prefetched data.
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 12 ++-
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h | 6 ++
llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp | 17 +++-
.../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 96 ++++++++++++-------
.../Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h | 16 ++--
5 files changed, 100 insertions(+), 47 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index e3bf41f8e513f..25cd2516e145c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -193,10 +193,13 @@ void AMDGPUAsmPrinter::emitFunctionBodyStart() {
const GCNSubtarget &STM = MF->getSubtarget<GCNSubtarget>();
const Function &F = MF->getFunction();
- // If ICache prefetch is enabled, create the function end symbol early so
- // it can be referenced by the prefetch MCExprs during instruction emission.
- if (MFI.hasICachePrefetch())
+ // If ICache prefetch is enabled, create the function end symbol and prefetch
+ // block start symbol early so it can be referenced by the prefetch MCExprs
+ // during instruction emission.
+ if (MFI.hasICachePrefetch()) {
PrefetchEndSym = createTempSymbol("pref_func_end");
+ PrefetchBlockStartSym = createTempSymbol("pref_block_start");
+ }
// TODO: We're checking this late, would be nice to check it earlier.
if (STM.requiresCodeObjectV6() && CodeObjectVersion < AMDGPU::AMDHSA_COV6) {
@@ -230,8 +233,9 @@ void AMDGPUAsmPrinter::emitFunctionBodyEnd() {
// This symbol was created in emitFunctionBodyStart for this function.
if (PrefetchEndSym) {
OutStreamer->emitLabel(PrefetchEndSym);
- PrefetchEndSym = nullptr;
}
+ PrefetchEndSym = nullptr;
+ PrefetchBlockStartSym = nullptr;
}
/// Set bits in a kernel descriptor MCExpr field:
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
index 871bd738b947b..22eb7da72812b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
@@ -60,6 +60,9 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
// Symbol for the function end, used by ICache prefetch MCExprs.
// Created early in emitFunctionBodyStart when prefetch is enabled.
MCSymbol *PrefetchEndSym = nullptr;
+ // Symbol for the first prefetch instruction, used by ICache prefetch MCExprs.
+ // Created early in emitFunctionBodyStart when prefetch is enabled.
+ MCSymbol *PrefetchBlockStartSym = nullptr;
// When appropriate, add a _dvgpr$ symbol.
void emitDVgprSymbol(MachineFunction &MF);
@@ -164,6 +167,9 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
/// Get the symbol for the function end, used for ICache prefetch MCExprs.
MCSymbol *getPrefetchEndSym() const { return PrefetchEndSym; }
+ /// Get the symbol for the first prefetch instruction, used for ICache
+ /// prefetch MCExprs.
+ MCSymbol *getPrefetchBlockStartSym() const { return PrefetchBlockStartSym; }
/// Get the code size estimate from SIProgramInfo.
uint64_t getCodeSize() { return CurrentProgramInfo.getFunctionCodeSize(*MF); }
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
index fd4651a41f8a8..be97f392d9bf4 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
@@ -477,9 +477,14 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
constexpr unsigned OffsetIdx = 0;
constexpr unsigned SdataIdx = 2;
- // The sdata operand contains the slot index (0-15) set by the pass.
+ // The sdata operand contains the slot index [0, N) set by the pass.
int64_t SlotIndex = TmpInst.getOperand(SdataIdx).getImm();
+ // If this is the first slot, emit the prefetch block start symbol
+ // before the instruction.
+ if (SlotIndex == 0)
+ OutStreamer->emitLabel(getPrefetchBlockStartSym());
+
// Create MCExpr for code size using label subtraction.
// This gives the exact code size at assembly time.
const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
@@ -490,12 +495,18 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
const MCExpr *SlotIndexExpr =
MCConstantExpr::create(SlotIndex, OutContext);
+ // Create MCExpr for the offset of the first prefetch instruction in the
+ // function.
+ const MCExpr *PrefetchBlockOffset = MCBinaryExpr::createSub(
+ MCSymbolRefExpr::create(getPrefetchBlockStartSym(), OutContext),
+ MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
+
// Create MCExprs that will be evaluated at fixup time when symbol
// positions are known.
const MCExpr *CachelinesExpr = AMDGPUMCExpr::createPrefetchCachelines(
- SlotIndexExpr, CodeSizeExpr, OutContext);
+ SlotIndexExpr, CodeSizeExpr, PrefetchBlockOffset, OutContext);
const MCExpr *OffsetExpr = AMDGPUMCExpr::createPrefetchOffset(
- SlotIndexExpr, CodeSizeExpr, OutContext);
+ SlotIndexExpr, CodeSizeExpr, PrefetchBlockOffset, OutContext);
// Replace the offset and sdata operands with MCExprs.
TmpInst.getOperand(OffsetIdx) = MCOperand::createExpr(OffsetExpr);
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index 9459f3e32ac1c..d349a2f29af11 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -68,10 +68,10 @@ unsigned AMDGPUMCExpr::getNumExpectedArgs(VariantKind Kind) {
return 1;
case AGVK_TotalNumVGPRs:
case AGVK_AlignTo:
- case AGVK_PrefetchCachelines:
- case AGVK_PrefetchOffset:
return 2;
case AGVK_ExtraSGPRs:
+ case AGVK_PrefetchCachelines:
+ case AGVK_PrefetchOffset:
return 3;
case AGVK_Occupancy:
return 9;
@@ -253,10 +253,34 @@ bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
return true;
}
+static uint64_t calcFirstPrefetchOffset(uint64_t PrefetchBlockOffset) {
+ constexpr unsigned CacheLineSize = 128;
+ // Start prefetching on the first cache line after the first prefetch
+ // instruction.
+ return llvm::alignDown(PrefetchBlockOffset, CacheLineSize) + CacheLineSize;
+}
+
+static uint64_t calcPrefetchSize(uint64_t CodeSizeInBytes,
+ uint64_t PrefetchBlockOffset) {
+ constexpr uint64_t MaxPrefetchSize = 64 * 1024; // 64KB ICache
+ uint64_t FirstPrefetch = calcFirstPrefetchOffset(PrefetchBlockOffset);
+ uint64_t ClampedCodeSize = std::min(CodeSizeInBytes, MaxPrefetchSize);
+ return ClampedCodeSize > FirstPrefetch ? ClampedCodeSize - FirstPrefetch : 0;
+}
+
+static uint64_t calcPrefetchSlotOffset(uint64_t SlotIndex) {
+ constexpr unsigned MaxCachelinesPerPrefetch = 32;
+ constexpr unsigned CacheLineSize = 128;
+ constexpr unsigned BytesPerPrefetch =
+ MaxCachelinesPerPrefetch * CacheLineSize;
+ return SlotIndex * BytesPerPrefetch;
+}
+
bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
const MCAssembler *Asm) const {
- uint64_t SlotIndex = 0, CodeSizeInBytes = 0;
- if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes}))
+ uint64_t SlotIndex = 0, CodeSizeInBytes = 0, PrefetchBlockOffset = 0;
+ if (!evaluateMCExprs(Args, Asm,
+ {SlotIndex, CodeSizeInBytes, PrefetchBlockOffset}))
return false;
// Constants for prefetch calculation.
@@ -264,17 +288,15 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
// one added). Cacheline size is 128 bytes. Each slot covers 4KiB.
constexpr unsigned MaxCachelinesPerPrefetch = 32;
constexpr unsigned CacheLineSize = 128;
- constexpr unsigned BytesPerPrefetch =
- MaxCachelinesPerPrefetch * CacheLineSize;
- constexpr uint64_t MaxPrefetchSize = 64 * 1024; // 64KB ICache
-
- // Clamp code size to maximum prefetchable size.
- uint64_t PrefetchSize = std::min(CodeSizeInBytes, MaxPrefetchSize);
+ uint64_t PrefetchSize =
+ calcPrefetchSize(CodeSizeInBytes, PrefetchBlockOffset);
// Calculate the byte offset for this slot.
- uint64_t SlotOffset = SlotIndex * BytesPerPrefetch;
+ uint64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
- // If this slot starts beyond the code size, return 0 (NOP).
+ // If this slot starts beyond the prefetchable region, use the minimum
+ // encoded prefetch size. evaluatePrefetchOffset() targets the instruction's
+ // own cache line for such slots.
if (SlotOffset >= PrefetchSize) {
Res = MCValue::get(static_cast<int64_t>(0));
return true;
@@ -295,29 +317,31 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
const MCAssembler *Asm) const {
- uint64_t SlotIndex = 0, CodeSizeInBytes = 0;
- if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes}))
+ uint64_t SlotIndex = 0, CodeSizeInBytes = 0, PrefetchBlockOffset = 0;
+ if (!evaluateMCExprs(Args, Asm,
+ {SlotIndex, CodeSizeInBytes, PrefetchBlockOffset}))
return false;
- constexpr unsigned MaxCachelinesPerPrefetch = 32;
- constexpr unsigned CacheLineSize = 128;
- constexpr uint64_t MaxPrefetchSize = 64 * 1024;
+ constexpr uint64_t PrefetchInstSize = 8;
- uint64_t PrefetchSize = std::min(CodeSizeInBytes, MaxPrefetchSize);
- uint64_t Offset = 0;
- uint64_t Remaining = PrefetchSize;
+ uint64_t FirstPrefetch = calcFirstPrefetchOffset(PrefetchBlockOffset);
+ uint64_t PrefetchSize =
+ calcPrefetchSize(CodeSizeInBytes, PrefetchBlockOffset);
- for (uint64_t I = 0; I < SlotIndex; ++I) {
- if (Remaining == 0)
- break;
- uint64_t Cachelines =
- std::min(divideCeil(Remaining, CacheLineSize),
- static_cast<uint64_t>(MaxCachelinesPerPrefetch));
- uint64_t PrefetchBytes = Cachelines * CacheLineSize;
- Offset += PrefetchBytes;
- Remaining = (Remaining > PrefetchBytes) ? Remaining - PrefetchBytes : 0;
+ uint64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
+ if (SlotOffset >= PrefetchSize) {
+ // Instruction semantics adds one to sdata when calculating the length of
+ // the prefetch. This means that even a prefetch instruction with sdata == 0
+ // still performs a prefetch. Therefore, to make this prefetch neutral, we
+ // let the prefetch instruction "prefetch" its own cache line.
+ // TODO: Check if we can replace this with proper s_nop instead.
+ Res = MCValue::get(static_cast<int64_t>(0));
+ return true;
}
-
+ // Prefetch is relative to this prefetch instruction's PC.
+ uint64_t PC = PrefetchBlockOffset + (SlotIndex * PrefetchInstSize);
+ uint64_t Target = FirstPrefetch + SlotOffset;
+ uint64_t Offset = Target - PC;
Res = MCValue::get(static_cast<int64_t>(Offset));
return true;
}
@@ -431,13 +455,17 @@ AMDGPUMCExpr::createInstPrefSize(const MCExpr *CodeSizeBytes, MCContext &Ctx) {
}
const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchCachelines(
- const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes, MCContext &Ctx) {
- return create(AGVK_PrefetchCachelines, {SlotIndex, CodeSizeBytes}, Ctx);
+ const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
+ const MCExpr *PrefetchBlockOffset, MCContext &Ctx) {
+ return create(AGVK_PrefetchCachelines,
+ {SlotIndex, CodeSizeBytes, PrefetchBlockOffset}, Ctx);
}
const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchOffset(
- const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes, MCContext &Ctx) {
- return create(AGVK_PrefetchOffset, {SlotIndex, CodeSizeBytes}, Ctx);
+ const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
+ const MCExpr *PrefetchBlockOffset, MCContext &Ctx) {
+ return create(AGVK_PrefetchOffset,
+ {SlotIndex, CodeSizeBytes, PrefetchBlockOffset}, Ctx);
}
const AMDGPUMCExpr *AMDGPUMCExpr::createLit(LitModifier Lit, int64_t Value,
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index d3693fb14045e..0db84e172f2ed 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -124,20 +124,24 @@ class AMDGPUMCExpr : public MCTargetExpr {
/// slot.
/// SlotIndex is the 0-based index of the prefetch instruction (0-15).
/// CodeSizeBytes is the total code size in bytes.
+ /// PrefetchBlockOffset is the byte offset of the first prefetch instruction
+ /// from the function entry.
/// Returns the requested cacheline count minus one, encoded for the 5-bit
/// sdata field.
static const AMDGPUMCExpr *
createPrefetchCachelines(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
- MCContext &Ctx);
+ const MCExpr *PrefetchBlockOffset, MCContext &Ctx);
/// Create an expression for computing the byte offset for a prefetch slot.
/// SlotIndex is the 0-based index of the prefetch instruction (0-15).
/// CodeSizeBytes is the total code size in bytes.
- /// Returns the cumulative byte offset where this slot should start
- /// prefetching.
- static const AMDGPUMCExpr *createPrefetchOffset(const MCExpr *SlotIndex,
- const MCExpr *CodeSizeBytes,
- MCContext &Ctx);
+ /// PrefetchBlockOffset is the byte offset of the first prefetch instruction
+ /// from the function entry.
+ /// Returns the byte offset from the PC of the corresponding prefetch
+ /// instruction to where this slot should start prefetching.
+ static const AMDGPUMCExpr *
+ createPrefetchOffset(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
+ const MCExpr *PrefetchBlockOffset, MCContext &Ctx);
static const AMDGPUMCExpr *createLit(LitModifier Lit, int64_t Value,
MCContext &Ctx);
>From 1e3a6075c53f09def8344c2b95ac70cedb39ce14 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Tue, 4 Aug 2026 02:42:18 -0500
Subject: [PATCH 06/22] Update icache prefetch test
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/test/CodeGen/AMDGPU/icache-prefetch.ll | 206 +++++++++++++-------
1 file changed, 139 insertions(+), 67 deletions(-)
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 2dc738cd703a6..45c2c6d08d0ce 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -8,105 +8,177 @@
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
; RUN: llvm-objdump -d %t.o | FileCheck -check-prefix=GFX1250-OBJ %s
-; Test that ICache prefetch instructions are inserted for entry functions on gfx1250.
-; Verify object file has resolved cacheline counts.
-; GFX1250-OBJ-LABEL: <entry_kernel>:
-; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x0, null, 2
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x100, null, 0
+; Use .space to make the estimated MachineFunction size and final assembled
+; size large without spelling out thousands of instructions.
-define amdgpu_kernel void @entry_kernel(ptr addrspace(1) %out) {
-; GFX1250-LABEL: entry_kernel:
+; A function below the 32 KiB kernel-descriptor prefetch limit must not use
+; explicit prefetch instructions.
+; GFX1250-OBJ-LABEL: <below_threshold>:
+; GFX1250-OBJ-NOT: s_prefetch_inst_pc_rel
+define amdgpu_kernel void @below_threshold() {
+; GFX1250-LABEL: below_threshold:
; GFX1250: ; %bb.0:
+; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: ;;#ASMSTART
+; GFX1250-NEXT: .space 30000
+; GFX1250-NEXT: ;;#ASMEND
+; GFX1250-NEXT: s_endpgm
+;
+; NO-PREFETCH-LABEL: below_threshold:
+; NO-PREFETCH: ; %bb.0:
+; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; NO-PREFETCH-NEXT: v_nop
+; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT: ;;#ASMSTART
+; NO-PREFETCH-NEXT: .space 30000
+; NO-PREFETCH-NEXT: ;;#ASMEND
+; NO-PREFETCH-NEXT: s_endpgm
+ call void asm sideeffect ".space 30000", ""()
+ ret void
+}
+
+; Exercise several full 4 KiB prefetch slots and a partial final slot.
+; GFX1250-OBJ-LABEL: <partial_final_slot>:
+; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x80, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x1078, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2070, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3068, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4060, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5058, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6050, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7048, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8040, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9038, null, 24
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x0, null, 0
+define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
+; GFX1250-LABEL: partial_final_slot:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: .Lpref_block_start0:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(0, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(1, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(2, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(3, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(4, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(5, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(6, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(7, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(8, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(9, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(10, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; GFX1250-NEXT: v_nop
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 24, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(0, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(1, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(2, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(3, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(4, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(5, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(6, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(7, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(8, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(9, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(10, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(11, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(12, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(13, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(14, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(15, .Lpref_func_end0-entry_kernel)
; GFX1250-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
; GFX1250-NEXT: v_mov_b32_e32 v0, 0
+; GFX1250-NEXT: ;;#ASMSTART
+; GFX1250-NEXT: .space 40000
+; GFX1250-NEXT: ;;#ASMEND
; GFX1250-NEXT: s_wait_kmcnt 0x0
; GFX1250-NEXT: global_store_b32 v0, v0, s[0:1]
; GFX1250-NEXT: s_endpgm
; GFX1250-NEXT: .Lpref_func_end0:
;
-; NO-PREFETCH-LABEL: entry_kernel:
+; NO-PREFETCH-LABEL: partial_final_slot:
; NO-PREFETCH: ; %bb.0:
+; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; NO-PREFETCH-NEXT: v_nop
; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
; NO-PREFETCH-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
; NO-PREFETCH-NEXT: v_mov_b32_e32 v0, 0
+; NO-PREFETCH-NEXT: ;;#ASMSTART
+; NO-PREFETCH-NEXT: .space 40000
+; NO-PREFETCH-NEXT: ;;#ASMEND
; NO-PREFETCH-NEXT: s_wait_kmcnt 0x0
; NO-PREFETCH-NEXT: global_store_b32 v0, v0, s[0:1]
; NO-PREFETCH-NEXT: s_endpgm
+ call void asm sideeffect ".space 40000", ""()
store i32 0, ptr addrspace(1) %out
ret void
}
-; Verify that called functions do NOT get prefetch instructions.
+; Exercise the 16-instruction limit and reserve the cache line containing the
+; first prefetch instruction instead of attempting to replace all 64 KiB.
+; GFX1250-OBJ-LABEL: <cache_size_limit>:
+; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x80, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x1078, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2070, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3068, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4060, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5058, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6050, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7048, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8040, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9038, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xa030, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xb028, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xc020, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xd018, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xe010, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xf008, null, 30
+define amdgpu_kernel void @cache_size_limit() {
+; GFX1250-LABEL: cache_size_limit:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: .Lpref_block_start1:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(0, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(1, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(2, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(3, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(4, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(5, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(6, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(7, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(8, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(9, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(10, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(11, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(12, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(13, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(14, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(15, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: ;;#ASMSTART
+; GFX1250-NEXT: .space 65536
+; GFX1250-NEXT: ;;#ASMEND
+; GFX1250-NEXT: s_endpgm
+; GFX1250-NEXT: .Lpref_func_end1:
+;
+; NO-PREFETCH-LABEL: cache_size_limit:
+; NO-PREFETCH: ; %bb.0:
+; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; NO-PREFETCH-NEXT: v_nop
+; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT: ;;#ASMSTART
+; NO-PREFETCH-NEXT: .space 65536
+; NO-PREFETCH-NEXT: ;;#ASMEND
+; NO-PREFETCH-NEXT: s_endpgm
+ call void asm sideeffect ".space 65536", ""()
+ ret void
+}
-define void @called_function(ptr addrspace(1) %out) {
+; Non-entry functions must not get explicit prefetch instructions regardless
+; of their size.
+define void @called_function() {
; GFX1250-LABEL: called_function:
; GFX1250: ; %bb.0:
; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-NEXT: s_wait_kmcnt 0x0
-; GFX1250-NEXT: v_mov_b32_e32 v2, 1
-; GFX1250-NEXT: global_store_b32 v[0:1], v2, off
+; GFX1250-NEXT: ;;#ASMSTART
+; GFX1250-NEXT: .space 40000
+; GFX1250-NEXT: ;;#ASMEND
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
;
; NO-PREFETCH-LABEL: called_function:
; NO-PREFETCH: ; %bb.0:
; NO-PREFETCH-NEXT: s_wait_loadcnt_dscnt 0x0
; NO-PREFETCH-NEXT: s_wait_kmcnt 0x0
-; NO-PREFETCH-NEXT: v_mov_b32_e32 v2, 1
-; NO-PREFETCH-NEXT: global_store_b32 v[0:1], v2, off
+; NO-PREFETCH-NEXT: ;;#ASMSTART
+; NO-PREFETCH-NEXT: .space 40000
+; NO-PREFETCH-NEXT: ;;#ASMEND
; NO-PREFETCH-NEXT: s_set_pc_i64 s[30:31]
- store i32 1, ptr addrspace(1) %out
- ret void
-}
-
-; GFX1250-OBJ-LABEL: <tiny_kernel>:
-; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x0, null, 2
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x100, null, 0
-
-define amdgpu_kernel void @tiny_kernel() {
-; GFX1250-LABEL: tiny_kernel:
-; GFX1250: ; %bb.0:
-; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 24, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(0, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(1, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(2, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(3, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(4, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(5, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(6, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(7, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(8, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(9, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(10, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(11, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(12, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(13, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(14, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(15, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT: s_endpgm
-; GFX1250-NEXT: .Lpref_func_end1:
-;
-; NO-PREFETCH-LABEL: tiny_kernel:
-; NO-PREFETCH: ; %bb.0:
-; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; NO-PREFETCH-NEXT: s_endpgm
+ call void asm sideeffect ".space 40000", ""()
ret void
}
>From 02f7aa5ffa52b47c85741d04c87cb11fb8f369a7 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Tue, 4 Aug 2026 03:39:21 -0500
Subject: [PATCH 07/22] Extract instruction prefetch into separate pass
Passes running after the pre-emit peephole pass can have a significant
impact on code size. Extract the instruction cache prefetching into a
separate pass, so it can run after those passes and insert based on a
more reliable code size estimation.
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/AMDGPU.h | 10 ++
.../AMDGPU/AMDGPUInsertICachePrefetch.cpp | 145 ++++++++++++++++++
llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def | 1 +
.../lib/Target/AMDGPU/AMDGPUTargetMachine.cpp | 7 +
llvm/lib/Target/AMDGPU/CMakeLists.txt | 1 +
llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp | 97 ------------
llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll | 2 +
llvm/test/CodeGen/AMDGPU/llc-pipeline.ll | 4 +
8 files changed, 170 insertions(+), 97 deletions(-)
create mode 100644 llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.h b/llvm/lib/Target/AMDGPU/AMDGPU.h
index 809540bd05b45..c117fe4083809 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.h
@@ -370,6 +370,13 @@ struct AMDGPUInsertDelayAluPass
MachineFunctionAnalysisManager &MFAM);
};
+class AMDGPUInsertICachePrefetchPass
+ : public RequiredPassInfoMixin<AMDGPUInsertICachePrefetchPass> {
+public:
+ PreservedAnalyses run(MachineFunction &MF,
+ MachineFunctionAnalysisManager &MFAM);
+};
+
FunctionPass *createAMDGPUISelDag(TargetMachine &TM, CodeGenOptLevel OptLevel);
ModulePass *createAMDGPUAlwaysInlinePass(bool GlobalOpt = true);
@@ -611,6 +618,9 @@ extern char &SIModeRegisterID;
void initializeAMDGPUInsertDelayAluLegacyPass(PassRegistry &);
extern char &AMDGPUInsertDelayAluID;
+void initializeAMDGPUInsertICachePrefetchLegacyPass(PassRegistry &);
+extern char &AMDGPUInsertICachePrefetchID;
+
void initializeAMDGPULowerVGPREncodingLegacyPass(PassRegistry &);
extern char &AMDGPULowerVGPREncodingLegacyID;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
new file mode 100644
index 0000000000000..119db87ac0a46
--- /dev/null
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -0,0 +1,145 @@
+//===- AMDGPUInsertICachePrefetch.cpp - Insert ICache prefetches ---------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+/// \file
+/// Insert instruction-cache prefetches for large AMDHSA entry functions.
+//
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPU.h"
+#include "GCNSubtarget.h"
+#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
+#include "SIMachineFunctionInfo.h"
+#include "SIProgramInfo.h"
+#include "llvm/CodeGen/MachineFunctionPass.h"
+#include "llvm/CodeGen/MachineInstrBuilder.h"
+#include "llvm/Support/CommandLine.h"
+#include "llvm/TargetParser/Triple.h"
+
+using namespace llvm;
+
+#define DEBUG_TYPE "amdgpu-insert-icache-prefetch"
+
+static cl::opt<bool>
+ EnableICachePrefetch("amdgpu-icache-prefetch",
+ cl::desc("Insert ICache prefetch instructions"),
+ cl::init(true), cl::Hidden);
+
+namespace {
+
+class AMDGPUInsertICachePrefetch {
+public:
+ bool run(MachineFunction &MF);
+};
+
+class AMDGPUInsertICachePrefetchLegacy : public MachineFunctionPass {
+public:
+ static char ID;
+
+ AMDGPUInsertICachePrefetchLegacy() : MachineFunctionPass(ID) {}
+
+ void getAnalysisUsage(AnalysisUsage &AU) const override {
+ AU.setPreservesCFG();
+ MachineFunctionPass::getAnalysisUsage(AU);
+ }
+
+ bool runOnMachineFunction(MachineFunction &MF) override {
+ return AMDGPUInsertICachePrefetch().run(MF);
+ }
+};
+
+} // end anonymous namespace
+
+bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
+ if (!EnableICachePrefetch)
+ return false;
+
+ const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
+ if (!ST.hasICachePrefetch())
+ return false;
+
+ // Only run for AMDHSA - this is where kernel descriptors are used and
+ // rsrc3 INST_PREF_SIZE is relevant.
+ if (ST.getTargetTriple().getOS() != Triple::AMDHSA)
+ return false;
+
+ SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
+ if (!MFI->isEntryFunction())
+ return false;
+
+ SIProgramInfo PI;
+ uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
+ // The kernel descriptor can specify an instruction prefetch size of up to 256
+ // in INST_PREF_SIZE. At a granularity of 128B, this equals 32KiB of
+ // instructions that can be prefetched without inserting explicit prefetch
+ // instructions.
+ constexpr uint64_t MaxKDPrefetch = 1u << 15;
+ if (ProgramSize <= MaxKDPrefetch)
+ return false;
+
+ MachineBasicBlock &EntryBB = MF.front();
+ MachineBasicBlock::iterator InsertPt = EntryBB.begin();
+
+ // Skip past any instructions that must remain at the very beginning:
+ // - Debug values and CFI instructions
+ // - S_SETREG_IMM32_B32 instructions that set up MODE register bits
+ // (e.g., REPLAY_MODE bit 25 from SIFrameLowering)
+ // We want the prefetches to come after all initial MODE setup.
+ while (InsertPt != EntryBB.end()) {
+ if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction() ||
+ InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32) {
+ ++InsertPt;
+ continue;
+ }
+ break;
+ }
+
+ const SIInstrInfo *TII = ST.getInstrInfo();
+ DebugLoc DL;
+
+ // Each prefetch can transfer 4KiB of instructions. Retain the existing
+ // slack for growth after this late pass, such as padding and alignment. The
+ // offset and sdata operands are placeholders; AMDGPUAsmPrinter fixes them
+ // up using the exact emitted code size.
+ constexpr uint64_t PrefetchSlack = 2 * 1024;
+ constexpr uint64_t BytesPerPrefetch = 4 * 1024;
+ // Each prefetch can transfer up to 32 cachelines of 128 bytes = 4KiB.
+ // 16 instructions cover 64KiB (the full ICache size).
+ constexpr unsigned MaxNumPrefetchInsts = 16;
+ unsigned NumPrefetches =
+ llvm::divideCeil(ProgramSize + PrefetchSlack, BytesPerPrefetch);
+ NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
+ for (unsigned I = 0; I < NumPrefetches; ++I) {
+ BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
+ .addImm(0) // offset (placeholder, fixed up later)
+ .addReg(AMDGPU::SGPR_NULL) // soffset
+ .addImm(I); // sdata (slot index, fixed up later)
+ }
+
+ // The instruction is fire-and-forget: it updates no wave wait counter and
+ // returns neither a result nor an error. It is therefore safe to insert
+ // after the wait-counter and hazard passes.
+ MFI->setHasICachePrefetch(true);
+ return true;
+}
+
+PreservedAnalyses
+llvm::AMDGPUInsertICachePrefetchPass::run(MachineFunction &MF,
+ MachineFunctionAnalysisManager &) {
+ if (!AMDGPUInsertICachePrefetch().run(MF))
+ return PreservedAnalyses::all();
+ auto PA = getMachineFunctionPassPreservedAnalyses();
+ PA.preserveSet<CFGAnalyses>();
+ return PA;
+}
+
+char AMDGPUInsertICachePrefetchLegacy::ID = 0;
+char &llvm::AMDGPUInsertICachePrefetchID = AMDGPUInsertICachePrefetchLegacy::ID;
+
+INITIALIZE_PASS(AMDGPUInsertICachePrefetchLegacy, DEBUG_TYPE,
+ "AMDGPU Insert ICache Prefetch", false, false)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def b/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def
index 372d5f5acab21..883dd4eab455b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def
@@ -119,6 +119,7 @@ MACHINE_FUNCTION_PASS("amdgpu-asm-printer", AMDGPUAsmPrinterPass())
MACHINE_FUNCTION_PASS("amdgpu-global-isel-divergence-lowering",
AMDGPUGlobalISelDivergenceLoweringPass())
MACHINE_FUNCTION_PASS("amdgpu-insert-delay-alu", AMDGPUInsertDelayAluPass())
+MACHINE_FUNCTION_PASS("amdgpu-insert-icache-prefetch", AMDGPUInsertICachePrefetchPass())
MACHINE_FUNCTION_PASS("amdgpu-isel", AMDGPUISelDAGToDAGPass(*this))
MACHINE_FUNCTION_PASS("amdgpu-lower-vgpr-encoding", AMDGPULowerVGPREncodingPass())
MACHINE_FUNCTION_PASS("amdgpu-mark-last-scratch-load", AMDGPUMarkLastScratchLoadPass())
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
index 72c028edbaef5..05a5f4545fcae 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
@@ -739,6 +739,7 @@ extern "C" LLVM_ABI LLVM_EXTERNAL_VISIBILITY void LLVMInitializeAMDGPUTarget() {
initializeAMDGPURewriteUndefForPHILegacyPass(*PR);
initializeSIAnnotateControlFlowLegacyPass(*PR);
initializeAMDGPUInsertDelayAluLegacyPass(*PR);
+ initializeAMDGPUInsertICachePrefetchLegacyPass(*PR);
initializeAMDGPULowerVGPREncodingLegacyPass(*PR);
initializeSIInsertHardClausesLegacyPass(*PR);
initializeSIInsertWaitcntsLegacyPass(*PR);
@@ -2066,6 +2067,9 @@ void GCNPassConfig::addPreEmitPass() {
if (isPassEnabled(EnableInsertDelayAlu, CodeGenOptLevel::Less))
addPass(&AMDGPUInsertDelayAluID);
+ if (getOptLevel() > CodeGenOptLevel::None)
+ addPass(&AMDGPUInsertICachePrefetchID);
+
addPass(&BranchRelaxationPassID);
}
@@ -2815,6 +2819,9 @@ void AMDGPUCodeGenPassBuilder::addPreEmitPass(PassManagerWrapper &PMW) {
addMachineFunctionPass(AMDGPUInsertDelayAluPass(), PMW);
}
+ if (TM.getOptLevel() > CodeGenOptLevel::None)
+ addMachineFunctionPass(AMDGPUInsertICachePrefetchPass(), PMW);
+
addMachineFunctionPass(BranchRelaxationPass(), PMW);
}
diff --git a/llvm/lib/Target/AMDGPU/CMakeLists.txt b/llvm/lib/Target/AMDGPU/CMakeLists.txt
index 4a5f77d55afa5..1bbdbc9ae2786 100644
--- a/llvm/lib/Target/AMDGPU/CMakeLists.txt
+++ b/llvm/lib/Target/AMDGPU/CMakeLists.txt
@@ -63,6 +63,7 @@ add_llvm_target(AMDGPUCodeGen
AMDGPUHSAMetadataStreamer.cpp
AMDGPUHWEvents.cpp
AMDGPUInsertDelayAlu.cpp
+ AMDGPUInsertICachePrefetch.cpp
AMDGPUInstCombineIntrinsic.cpp
AMDGPUUniformIntrinsicCombine.cpp
AMDGPUInstrInfo.cpp
diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index 11344ff8e4709..e5b1de7bc4ec9 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -20,9 +20,6 @@
#include "AMDGPU.h"
#include "GCNSubtarget.h"
-#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
-#include "SIMachineFunctionInfo.h"
-#include "SIProgramInfo.h"
#include "llvm/ADT/Statistic.h"
#include "llvm/CodeGen/MachineDominators.h"
#include "llvm/CodeGen/MachineFunctionPass.h"
@@ -30,8 +27,6 @@
#include "llvm/CodeGen/MachineLoopInfo.h"
#include "llvm/CodeGen/TargetSchedule.h"
#include "llvm/Support/BranchProbability.h"
-#include "llvm/Support/CommandLine.h"
-#include "llvm/TargetParser/Triple.h"
using namespace llvm;
#define DEBUG_TYPE "si-pre-emit-peephole"
@@ -51,13 +46,6 @@ struct ModeFieldState {
bool isTracked() const { return PendingWrite || Value; }
};
-static cl::opt<bool>
- EnableICachePrefetch("amdgpu-icache-prefetch",
- cl::desc("Insert ICache prefetch instructions"),
- cl::init(true), cl::Hidden);
-
-namespace {
-
class SIPreEmitPeephole {
private:
const GCNSubtarget *ST = nullptr;
@@ -65,7 +53,6 @@ class SIPreEmitPeephole {
const SIRegisterInfo *TRI = nullptr;
MachineLoopInfo *MLI = nullptr;
- bool insertICachePrefetch(MachineFunction &MF);
bool optimizeVccBranch(MachineInstr &MI) const;
void updateMLIBeforeRemovingEdge(MachineBasicBlock *From,
MachineBasicBlock *To) const;
@@ -864,86 +851,6 @@ MachineInstrBuilder SIPreEmitPeephole::createUnpackedMI(MachineInstr &I,
return NewMI;
}
-bool SIPreEmitPeephole::insertICachePrefetch(MachineFunction &MF) {
- // Only run on targets that support ICache prefetching.
- if (!ST->hasICachePrefetch())
- return false;
-
- // Only run for AMDHSA - this is where kernel descriptors are used
- // and rsrc3 INST_PREF_SIZE is relevant.
- const Triple &TT = ST->getTargetTriple();
- if (TT.getOS() != Triple::AMDHSA)
- return false;
-
- SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
-
- // Only insert prefetch instructions for entry functions.
- if (!MFI->isEntryFunction())
- return false;
-
- SIProgramInfo PI;
- uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
- // The kernel descriptor can specify an instruction prefetch size of up to 256
- // in INST_PREF_SIZE. At a granularity of 128B, this equals 32KiB of
- // instructions that can be prefetched without inserting explicit prefetch
- // instructions.
- constexpr uint64_t MaxKDPrefetch = 1u << 15;
- if (ProgramSize <= MaxKDPrefetch)
- return false;
-
- MachineBasicBlock &EntryBB = MF.front();
- MachineBasicBlock::iterator InsertPt = EntryBB.begin();
-
- // Skip past any instructions that must remain at the very beginning:
- // - Debug values and CFI instructions
- // - S_SETREG_IMM32_B32 instructions that set up MODE register bits
- // (e.g., REPLAY_MODE bit 25 from SIFrameLowering)
- // We want the prefetches to come after all initial MODE setup.
- while (InsertPt != EntryBB.end()) {
- if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction()) {
- ++InsertPt;
- continue;
- }
- if (InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32) {
- ++InsertPt;
- continue;
- }
- break;
- }
-
- DebugLoc DL;
-
- // Calculate the number of prefetch instructions required for the current
- // program size. Each prefetch can transfer 4KiB of instructions. Add
- // some slack as inserting the prefetches and later transformations, e.g.,
- // padding and alignment, will introduce additional bytes. The offset and
- // sdata operands are placeholders - the sdata operand stores the slot index
- // (0-15). Both will be fixed up in AMDGPUAsmPrinter based on the actual code
- // size.
- constexpr uint64_t PrefetchSlack = 2 * 1024;
- constexpr uint64_t BytesPerPrefetch = 4 * 1024;
- // Each prefetch can transfer up to 32 cachelines of 128 bytes = 4KiB.
- // 16 instructions cover 64KiB (the full ICache size).
- constexpr unsigned MaxNumPrefetchInsts = 16;
- unsigned NumPrefetches =
- llvm::divideCeil(ProgramSize + PrefetchSlack, BytesPerPrefetch);
- // Limit to 16 prefetches at most, otherwise we'd exceed the cache size.
- NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
- for (unsigned I = 0; I < NumPrefetches; ++I) {
- BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
- .addImm(0) // offset (placeholder, fixed up later)
- .addReg(AMDGPU::SGPR_NULL) // soffset
- .addImm(I); // sdata (slot index, fixed up later)
- }
-
- // Mark that we've inserted ICache prefetch instructions.
- // This tells AsmPrinter to set rsrc3 INST_PREF_SIZE to 1 and fix up
- // the prefetch cacheline counts.
- MFI->setHasICachePrefetch(true);
-
- return true;
-}
-
PreservedAnalyses
llvm::SIPreEmitPeepholePass::run(MachineFunction &MF,
MachineFunctionAnalysisManager &MFAM) {
@@ -966,10 +873,6 @@ bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
MLI = LoopInfo;
bool Changed = false;
- // Insert ICache prefetch instructions if enabled.
- if (EnableICachePrefetch)
- Changed |= insertICachePrefetch(MF);
-
MF.RenumberBlocks();
for (MachineBasicBlock &MBB : MF) {
diff --git a/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll b/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll
index cacab31f263c8..56b61ac7ea90d 100644
--- a/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll
+++ b/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll
@@ -284,6 +284,7 @@
; GCN-O2-NEXT: amdgpu-wait-sgpr-hazards
; GCN-O2-NEXT: amdgpu-lower-vgpr-encoding
; GCN-O2-NEXT: amdgpu-insert-delay-alu
+; GCN-O2-NEXT: amdgpu-insert-icache-prefetch
; GCN-O2-NEXT: branch-relaxation
; GCN-O2-NEXT: reg-usage-collector
; GCN-O2-NEXT: remove-loads-into-fake-uses
@@ -473,6 +474,7 @@
; GCN-O3-NEXT: amdgpu-wait-sgpr-hazards
; GCN-O3-NEXT: amdgpu-lower-vgpr-encoding
; GCN-O3-NEXT: amdgpu-insert-delay-alu
+; GCN-O3-NEXT: amdgpu-insert-icache-prefetch
; GCN-O3-NEXT: branch-relaxation
; GCN-O3-NEXT: reg-usage-collector
; GCN-O3-NEXT: remove-loads-into-fake-uses
diff --git a/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll b/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
index 59c74e1b18c9c..8f88d84280b47 100644
--- a/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
+++ b/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
@@ -455,6 +455,7 @@
; GCN-O1-NEXT: AMDGPU Insert waits for SGPR read hazards
; GCN-O1-NEXT: AMDGPU Lower VGPR Encoding
; GCN-O1-NEXT: AMDGPU Insert Delay ALU
+; GCN-O1-NEXT: AMDGPU Insert ICache Prefetch
; GCN-O1-NEXT: Branch relaxation pass
; GCN-O1-NEXT: Register Usage Information Collector Pass
; GCN-O1-NEXT: Remove Loads Into Fake Uses
@@ -785,6 +786,7 @@
; GCN-O1-OPTS-NEXT: AMDGPU Insert waits for SGPR read hazards
; GCN-O1-OPTS-NEXT: AMDGPU Lower VGPR Encoding
; GCN-O1-OPTS-NEXT: AMDGPU Insert Delay ALU
+; GCN-O1-OPTS-NEXT: AMDGPU Insert ICache Prefetch
; GCN-O1-OPTS-NEXT: Branch relaxation pass
; GCN-O1-OPTS-NEXT: Register Usage Information Collector Pass
; GCN-O1-OPTS-NEXT: Remove Loads Into Fake Uses
@@ -1120,6 +1122,7 @@
; GCN-O2-NEXT: AMDGPU Insert waits for SGPR read hazards
; GCN-O2-NEXT: AMDGPU Lower VGPR Encoding
; GCN-O2-NEXT: AMDGPU Insert Delay ALU
+; GCN-O2-NEXT: AMDGPU Insert ICache Prefetch
; GCN-O2-NEXT: Branch relaxation pass
; GCN-O2-NEXT: Register Usage Information Collector Pass
; GCN-O2-NEXT: Remove Loads Into Fake Uses
@@ -1470,6 +1473,7 @@
; GCN-O3-NEXT: AMDGPU Insert waits for SGPR read hazards
; GCN-O3-NEXT: AMDGPU Lower VGPR Encoding
; GCN-O3-NEXT: AMDGPU Insert Delay ALU
+; GCN-O3-NEXT: AMDGPU Insert ICache Prefetch
; GCN-O3-NEXT: Branch relaxation pass
; GCN-O3-NEXT: Register Usage Information Collector Pass
; GCN-O3-NEXT: Remove Loads Into Fake Uses
>From 3dc1282ecc4c4c87421fa24287f27d2c18345890 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Wed, 5 Aug 2026 02:21:36 -0500
Subject: [PATCH 08/22] Skip unclaused VMEM prologue for insertion
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
.../AMDGPU/AMDGPUInsertICachePrefetch.cpp | 14 ++++
llvm/test/CodeGen/AMDGPU/icache-prefetch.ll | 64 +++++++++----------
2 files changed, 46 insertions(+), 32 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index 119db87ac0a46..213fce4295d7f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -87,15 +87,29 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
// Skip past any instructions that must remain at the very beginning:
// - Debug values and CFI instructions
+ // - The gfx1250 initial unclaused-VMEM workaround
// - S_SETREG_IMM32_B32 instructions that set up MODE register bits
// (e.g., REPLAY_MODE bit 25 from SIFrameLowering)
// We want the prefetches to come after all initial MODE setup.
+ bool SkippedInitialUnclausedVmemPrologue =
+ !ST.hasRequiresInitialUnclausedVmem();
while (InsertPt != EntryBB.end()) {
if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction() ||
InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32) {
++InsertPt;
continue;
}
+ if (!SkippedInitialUnclausedVmemPrologue &&
+ InsertPt->getOpcode() == AMDGPU::GLOBAL_PREFETCH_B8_SADDR) {
+ auto Next = InsertPt;
+ ++Next;
+ if (Next != EntryBB.end() &&
+ Next->getOpcode() == AMDGPU::V_NOP_e32) {
+ InsertPt = ++Next;
+ SkippedInitialUnclausedVmemPrologue = true;
+ continue;
+ }
+ }
break;
}
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 45c2c6d08d0ce..18777c160cc4e 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -41,20 +41,23 @@ define amdgpu_kernel void @below_threshold() {
; Exercise several full 4 KiB prefetch slots and a partial final slot.
; GFX1250-OBJ-LABEL: <partial_final_slot>:
-; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x80, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x1078, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2070, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3068, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4060, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5058, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6050, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7048, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8040, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9038, null, 24
+; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x68, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x1060, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2058, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3050, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4048, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5040, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6038, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7030, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8028, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9020, null, 24
; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x0, null, 0
define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; GFX1250-LABEL: partial_final_slot:
; GFX1250: ; %bb.0:
+; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
; GFX1250-NEXT: .Lpref_block_start0:
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(0, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(1, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
@@ -67,9 +70,6 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(8, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(9, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(10, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; GFX1250-NEXT: v_nop
-; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
; GFX1250-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
; GFX1250-NEXT: v_mov_b32_e32 v0, 0
; GFX1250-NEXT: ;;#ASMSTART
@@ -101,25 +101,28 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; Exercise the 16-instruction limit and reserve the cache line containing the
; first prefetch instruction instead of attempting to replace all 64 KiB.
; GFX1250-OBJ-LABEL: <cache_size_limit>:
-; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x80, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x1078, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2070, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3068, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4060, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5058, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6050, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7048, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8040, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9038, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xa030, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xb028, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xc020, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xd018, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xe010, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xf008, null, 30
+; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x68, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x1060, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2058, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3050, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4048, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5040, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6038, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7030, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8028, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9020, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xa018, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xb010, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xc008, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xd000, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xdff8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xeff0, null, 30
define amdgpu_kernel void @cache_size_limit() {
; GFX1250-LABEL: cache_size_limit:
; GFX1250: ; %bb.0:
+; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
; GFX1250-NEXT: .Lpref_block_start1:
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(0, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(1, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
@@ -137,9 +140,6 @@ define amdgpu_kernel void @cache_size_limit() {
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(13, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(14, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(15, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; GFX1250-NEXT: v_nop
-; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
; GFX1250-NEXT: ;;#ASMSTART
; GFX1250-NEXT: .space 65536
; GFX1250-NEXT: ;;#ASMEND
>From 0602d751240c3c66fbfa027aa1f7bff70ced701e Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 6 Aug 2026 13:55:38 -0500
Subject: [PATCH 09/22] Distribute prefetch instructions across CFG
Instead of inserting all prefetch instructions in the entry block,
distribute them across the entry block and its post-domination chain.
Each block will only prefetch as much code as is needed before the next
candidate, plus some slack to account for latency.
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 10 +-
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h | 6 -
.../AMDGPU/AMDGPUInsertICachePrefetch.cpp | 170 +++++++++---
llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp | 18 +-
.../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 65 ++---
.../Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h | 17 +-
llvm/lib/Target/AMDGPU/SIProgramInfo.cpp | 23 +-
llvm/lib/Target/AMDGPU/SIProgramInfo.h | 6 +
llvm/test/CodeGen/AMDGPU/icache-prefetch.ll | 257 ++++++++++++++----
9 files changed, 395 insertions(+), 177 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 25cd2516e145c..fcbc576364e3f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -193,13 +193,10 @@ void AMDGPUAsmPrinter::emitFunctionBodyStart() {
const GCNSubtarget &STM = MF->getSubtarget<GCNSubtarget>();
const Function &F = MF->getFunction();
- // If ICache prefetch is enabled, create the function end symbol and prefetch
- // block start symbol early so it can be referenced by the prefetch MCExprs
- // during instruction emission.
- if (MFI.hasICachePrefetch()) {
+ // If ICache prefetch is enabled, create the function end symbol early so it
+ // can be referenced by the prefetch MCExprs during instruction emission.
+ if (MFI.hasICachePrefetch())
PrefetchEndSym = createTempSymbol("pref_func_end");
- PrefetchBlockStartSym = createTempSymbol("pref_block_start");
- }
// TODO: We're checking this late, would be nice to check it earlier.
if (STM.requiresCodeObjectV6() && CodeObjectVersion < AMDGPU::AMDHSA_COV6) {
@@ -235,7 +232,6 @@ void AMDGPUAsmPrinter::emitFunctionBodyEnd() {
OutStreamer->emitLabel(PrefetchEndSym);
}
PrefetchEndSym = nullptr;
- PrefetchBlockStartSym = nullptr;
}
/// Set bits in a kernel descriptor MCExpr field:
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
index 22eb7da72812b..871bd738b947b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
@@ -60,9 +60,6 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
// Symbol for the function end, used by ICache prefetch MCExprs.
// Created early in emitFunctionBodyStart when prefetch is enabled.
MCSymbol *PrefetchEndSym = nullptr;
- // Symbol for the first prefetch instruction, used by ICache prefetch MCExprs.
- // Created early in emitFunctionBodyStart when prefetch is enabled.
- MCSymbol *PrefetchBlockStartSym = nullptr;
// When appropriate, add a _dvgpr$ symbol.
void emitDVgprSymbol(MachineFunction &MF);
@@ -167,9 +164,6 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
/// Get the symbol for the function end, used for ICache prefetch MCExprs.
MCSymbol *getPrefetchEndSym() const { return PrefetchEndSym; }
- /// Get the symbol for the first prefetch instruction, used for ICache
- /// prefetch MCExprs.
- MCSymbol *getPrefetchBlockStartSym() const { return PrefetchBlockStartSym; }
/// Get the code size estimate from SIProgramInfo.
uint64_t getCodeSize() { return CurrentProgramInfo.getFunctionCodeSize(*MF); }
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index 213fce4295d7f..be8703c4c9d6c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -9,6 +9,10 @@
/// \file
/// Insert instruction-cache prefetches for large AMDHSA entry functions.
//
+// The prefetch instruction is fire-and-forget: it updates no wave wait counter
+// and returns neither a result nor an error. It is therefore safe to insert
+// after the wait-counter and hazard passes.
+//
//===----------------------------------------------------------------------===//
#include "AMDGPU.h"
@@ -16,8 +20,12 @@
#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
#include "SIMachineFunctionInfo.h"
#include "SIProgramInfo.h"
+#include "llvm/ADT/DenseMap.h"
#include "llvm/CodeGen/MachineFunctionPass.h"
#include "llvm/CodeGen/MachineInstrBuilder.h"
+#include "llvm/CodeGen/MachineLoopInfo.h"
+#include "llvm/CodeGen/MachinePostDominators.h"
+#include "llvm/InitializePasses.h"
#include "llvm/Support/CommandLine.h"
#include "llvm/TargetParser/Triple.h"
@@ -33,7 +41,22 @@ static cl::opt<bool>
namespace {
class AMDGPUInsertICachePrefetch {
+ MachineLoopInfo &MLI;
+ MachinePostDominatorTree &PDT;
+
+ bool isLoopFreeEntryPostDominator(const MachineBasicBlock &MBB,
+ const MachineBasicBlock &EntryBB) const {
+ return !MLI.getLoopFor(&MBB) && PDT.dominates(&MBB, &EntryBB);
+ }
+
public:
+ // These analyses describe the final machine CFG at this late insertion
+ // point. CFG-based placement will use them to select loop-free blocks on
+ // the entry block's post-dominator chain.
+ AMDGPUInsertICachePrefetch(MachineLoopInfo &MLI,
+ MachinePostDominatorTree &PDT)
+ : MLI(MLI), PDT(PDT) {}
+
bool run(MachineFunction &MF);
};
@@ -45,16 +68,57 @@ class AMDGPUInsertICachePrefetchLegacy : public MachineFunctionPass {
void getAnalysisUsage(AnalysisUsage &AU) const override {
AU.setPreservesCFG();
+ AU.addRequired<MachineLoopInfoWrapperPass>();
+ AU.addRequired<MachinePostDominatorTreeWrapperPass>();
MachineFunctionPass::getAnalysisUsage(AU);
}
bool runOnMachineFunction(MachineFunction &MF) override {
- return AMDGPUInsertICachePrefetch().run(MF);
+ auto &MLI = getAnalysis<MachineLoopInfoWrapperPass>().getLI();
+ auto &PDT =
+ getAnalysis<MachinePostDominatorTreeWrapperPass>().getPostDomTree();
+ return AMDGPUInsertICachePrefetch(MLI, PDT).run(MF);
}
};
} // end anonymous namespace
+static MachineBasicBlock::iterator
+findMBBInsertionPoint(MachineBasicBlock &MBB, const GCNSubtarget &ST,
+ bool IsEntryBlock) {
+ MachineBasicBlock::iterator InsertPt = MBB.begin();
+
+ // Skip past any instructions that must remain at the very beginning:
+ // - Debug values and CFI instructions
+ // - In the entry block only, the gfx1250 initial unclaused-VMEM workaround
+ // and S_SETREG_IMM32_B32 instructions that set up MODE register bits
+ // (e.g., REPLAY_MODE bit 25 from SIFrameLowering)
+ // In the entry block, we want the prefetches to come after all initial MODE
+ // setup.
+ bool SkippedInitialUnclausedVmemPrologue =
+ !IsEntryBlock || !ST.hasRequiresInitialUnclausedVmem();
+ while (InsertPt != MBB.end()) {
+ if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction() ||
+ (IsEntryBlock &&
+ InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32)) {
+ ++InsertPt;
+ continue;
+ }
+ if (!SkippedInitialUnclausedVmemPrologue &&
+ InsertPt->getOpcode() == AMDGPU::GLOBAL_PREFETCH_B8_SADDR) {
+ auto Next = InsertPt;
+ ++Next;
+ if (Next != MBB.end() && Next->getOpcode() == AMDGPU::V_NOP_e32) {
+ InsertPt = ++Next;
+ SkippedInitialUnclausedVmemPrologue = true;
+ continue;
+ }
+ }
+ break;
+ }
+ return InsertPt;
+}
+
bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
if (!EnableICachePrefetch)
return false;
@@ -82,38 +146,34 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
if (ProgramSize <= MaxKDPrefetch)
return false;
+ const SIInstrInfo *TII = ST.getInstrInfo();
MachineBasicBlock &EntryBB = MF.front();
- MachineBasicBlock::iterator InsertPt = EntryBB.begin();
- // Skip past any instructions that must remain at the very beginning:
- // - Debug values and CFI instructions
- // - The gfx1250 initial unclaused-VMEM workaround
- // - S_SETREG_IMM32_B32 instructions that set up MODE register bits
- // (e.g., REPLAY_MODE bit 25 from SIFrameLowering)
- // We want the prefetches to come after all initial MODE setup.
- bool SkippedInitialUnclausedVmemPrologue =
- !ST.hasRequiresInitialUnclausedVmem();
- while (InsertPt != EntryBB.end()) {
- if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction() ||
- InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32) {
- ++InsertPt;
+ // Walk the post-dominator chain to get candidates in execution order. This
+ // is distinct from the layout order used below to calculate code offsets.
+ SmallVector<MachineBasicBlock *> Candidates = {&EntryBB};
+ for (auto *Node = PDT.getNode(&EntryBB); Node; Node = Node->getIDom()) {
+ MachineBasicBlock *CandBB = Node->getBlock();
+ if (!CandBB)
+ break;
+ if (CandBB == &EntryBB ||
+ !isLoopFreeEntryPostDominator(*CandBB, EntryBB))
continue;
- }
- if (!SkippedInitialUnclausedVmemPrologue &&
- InsertPt->getOpcode() == AMDGPU::GLOBAL_PREFETCH_B8_SADDR) {
- auto Next = InsertPt;
- ++Next;
- if (Next != EntryBB.end() &&
- Next->getOpcode() == AMDGPU::V_NOP_e32) {
- InsertPt = ++Next;
- SkippedInitialUnclausedVmemPrologue = true;
- continue;
- }
- }
- break;
+ Candidates.push_back(CandBB);
+ }
+
+ // Record each candidate's current layout offset. This is the order in which
+ // the assembler emits blocks, and is used to determine the code range
+ // covered by a prefetch.
+ DenseMap<MachineBasicBlock *, uint64_t> CandidateOffsets;
+ uint64_t CodeSize = 0;
+ for (MachineBasicBlock &MB : MF) {
+ CodeSize = alignTo(CodeSize, MB.getAlignment());
+ if (isLoopFreeEntryPostDominator(MB, EntryBB))
+ CandidateOffsets[&MB] = CodeSize;
+ CodeSize += SIProgramInfo::getMachineBasicBlockCodeSize(MB, *TII);
}
- const SIInstrInfo *TII = ST.getInstrInfo();
DebugLoc DL;
// Each prefetch can transfer 4KiB of instructions. Retain the existing
@@ -128,24 +188,46 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
unsigned NumPrefetches =
llvm::divideCeil(ProgramSize + PrefetchSlack, BytesPerPrefetch);
NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
- for (unsigned I = 0; I < NumPrefetches; ++I) {
- BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
- .addImm(0) // offset (placeholder, fixed up later)
- .addReg(AMDGPU::SGPR_NULL) // soffset
- .addImm(I); // sdata (slot index, fixed up later)
+
+ size_t NumCandidates = Candidates.size();
+ unsigned Prefetches = 0;
+ // In each candidate block, prefetch as much code as necessary before control
+ // flow reaches the next candidate block, plus some slack to account for
+ // prefetch latency.
+ for (size_t Cand = 0, NextCand = 1; Cand < NumCandidates;
+ ++Cand, ++NextCand) {
+ unsigned PrefetchBeforeNext = NumPrefetches;
+ if (NextCand < NumCandidates) {
+ // To the offset of the next candidate we add:
+ // - PrefetchSlack: To make sure the last prefetch has the correct number
+ // of cache lines.
+ // - BytesPerPrefetch: To account for the latency of the prefetch.
+ unsigned PrefetchesBeforeNext = llvm::divideCeil(
+ CandidateOffsets.lookup(Candidates[NextCand]) + PrefetchSlack +
+ BytesPerPrefetch,
+ BytesPerPrefetch);
+ PrefetchBeforeNext = std::min(PrefetchesBeforeNext, PrefetchBeforeNext);
+ }
+ MachineBasicBlock *CandBB = Candidates[Cand];
+ MachineBasicBlock::iterator InsertPt =
+ findMBBInsertionPoint(*CandBB, ST, CandBB == &EntryBB);
+ for (; Prefetches < PrefetchBeforeNext; ++Prefetches) {
+ BuildMI(*CandBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
+ .addImm(0) // offset (placeholder, fixed up later)
+ .addReg(AMDGPU::SGPR_NULL) // soffset
+ .addImm(Prefetches); // sdata (slot index, fixed up later)
+ }
}
- // The instruction is fire-and-forget: it updates no wave wait counter and
- // returns neither a result nor an error. It is therefore safe to insert
- // after the wait-counter and hazard passes.
MFI->setHasICachePrefetch(true);
return true;
}
-PreservedAnalyses
-llvm::AMDGPUInsertICachePrefetchPass::run(MachineFunction &MF,
- MachineFunctionAnalysisManager &) {
- if (!AMDGPUInsertICachePrefetch().run(MF))
+PreservedAnalyses llvm::AMDGPUInsertICachePrefetchPass::run(
+ MachineFunction &MF, MachineFunctionAnalysisManager &MFAM) {
+ auto &MLI = MFAM.getResult<MachineLoopAnalysis>(MF);
+ auto &PDT = MFAM.getResult<MachinePostDominatorTreeAnalysis>(MF);
+ if (!AMDGPUInsertICachePrefetch(MLI, PDT).run(MF))
return PreservedAnalyses::all();
auto PA = getMachineFunctionPassPreservedAnalyses();
PA.preserveSet<CFGAnalyses>();
@@ -155,5 +237,9 @@ llvm::AMDGPUInsertICachePrefetchPass::run(MachineFunction &MF,
char AMDGPUInsertICachePrefetchLegacy::ID = 0;
char &llvm::AMDGPUInsertICachePrefetchID = AMDGPUInsertICachePrefetchLegacy::ID;
-INITIALIZE_PASS(AMDGPUInsertICachePrefetchLegacy, DEBUG_TYPE,
- "AMDGPU Insert ICache Prefetch", false, false)
+INITIALIZE_PASS_BEGIN(AMDGPUInsertICachePrefetchLegacy, DEBUG_TYPE,
+ "AMDGPU Insert ICache Prefetch", false, false)
+INITIALIZE_PASS_DEPENDENCY(MachineLoopInfoWrapperPass)
+INITIALIZE_PASS_DEPENDENCY(MachinePostDominatorTreeWrapperPass)
+INITIALIZE_PASS_END(AMDGPUInsertICachePrefetchLegacy, DEBUG_TYPE,
+ "AMDGPU Insert ICache Prefetch", false, false)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
index be97f392d9bf4..7e8f53bee9db6 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
@@ -480,10 +480,10 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
// The sdata operand contains the slot index [0, N) set by the pass.
int64_t SlotIndex = TmpInst.getOperand(SdataIdx).getImm();
- // If this is the first slot, emit the prefetch block start symbol
- // before the instruction.
- if (SlotIndex == 0)
- OutStreamer->emitLabel(getPrefetchBlockStartSym());
+ // Emit a symbol for each prefetch instruction to calculate the offset.
+ MCSymbol *InstOffsetSym =
+ createTempSymbol("pref_inst_offset_" + Twine(SlotIndex));
+ OutStreamer->emitLabel(InstOffsetSym);
// Create MCExpr for code size using label subtraction.
// This gives the exact code size at assembly time.
@@ -495,18 +495,18 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
const MCExpr *SlotIndexExpr =
MCConstantExpr::create(SlotIndex, OutContext);
- // Create MCExpr for the offset of the first prefetch instruction in the
+ // Create an MCExpr for this prefetch instruction's offset in the
// function.
- const MCExpr *PrefetchBlockOffset = MCBinaryExpr::createSub(
- MCSymbolRefExpr::create(getPrefetchBlockStartSym(), OutContext),
+ const MCExpr *PrefetchInstOffset = MCBinaryExpr::createSub(
+ MCSymbolRefExpr::create(InstOffsetSym, OutContext),
MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
// Create MCExprs that will be evaluated at fixup time when symbol
// positions are known.
const MCExpr *CachelinesExpr = AMDGPUMCExpr::createPrefetchCachelines(
- SlotIndexExpr, CodeSizeExpr, PrefetchBlockOffset, OutContext);
+ SlotIndexExpr, CodeSizeExpr, PrefetchInstOffset, OutContext);
const MCExpr *OffsetExpr = AMDGPUMCExpr::createPrefetchOffset(
- SlotIndexExpr, CodeSizeExpr, PrefetchBlockOffset, OutContext);
+ SlotIndexExpr, CodeSizeExpr, PrefetchInstOffset, OutContext);
// Replace the offset and sdata operands with MCExprs.
TmpInst.getOperand(OffsetIdx) = MCOperand::createExpr(OffsetExpr);
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index d349a2f29af11..f0bbab45f6e06 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -253,19 +253,9 @@ bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
return true;
}
-static uint64_t calcFirstPrefetchOffset(uint64_t PrefetchBlockOffset) {
- constexpr unsigned CacheLineSize = 128;
- // Start prefetching on the first cache line after the first prefetch
- // instruction.
- return llvm::alignDown(PrefetchBlockOffset, CacheLineSize) + CacheLineSize;
-}
-
-static uint64_t calcPrefetchSize(uint64_t CodeSizeInBytes,
- uint64_t PrefetchBlockOffset) {
+static uint64_t calcPrefetchSize(uint64_t CodeSizeInBytes) {
constexpr uint64_t MaxPrefetchSize = 64 * 1024; // 64KB ICache
- uint64_t FirstPrefetch = calcFirstPrefetchOffset(PrefetchBlockOffset);
- uint64_t ClampedCodeSize = std::min(CodeSizeInBytes, MaxPrefetchSize);
- return ClampedCodeSize > FirstPrefetch ? ClampedCodeSize - FirstPrefetch : 0;
+ return std::min(CodeSizeInBytes, MaxPrefetchSize);
}
static uint64_t calcPrefetchSlotOffset(uint64_t SlotIndex) {
@@ -278,18 +268,16 @@ static uint64_t calcPrefetchSlotOffset(uint64_t SlotIndex) {
bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
const MCAssembler *Asm) const {
- uint64_t SlotIndex = 0, CodeSizeInBytes = 0, PrefetchBlockOffset = 0;
- if (!evaluateMCExprs(Args, Asm,
- {SlotIndex, CodeSizeInBytes, PrefetchBlockOffset}))
+ uint64_t SlotIndex = 0, CodeSizeInBytes = 0, InstOffset = 0;
+ if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes, InstOffset}))
return false;
// Constants for prefetch calculation.
// Each instruction can prefetch up to 32 cachelines (5-bit sdata field, plus
// one added). Cacheline size is 128 bytes. Each slot covers 4KiB.
- constexpr unsigned MaxCachelinesPerPrefetch = 32;
+ constexpr uint64_t MaxCachelinesPerPrefetch = 32;
constexpr unsigned CacheLineSize = 128;
- uint64_t PrefetchSize =
- calcPrefetchSize(CodeSizeInBytes, PrefetchBlockOffset);
+ uint64_t PrefetchSize = calcPrefetchSize(CodeSizeInBytes);
// Calculate the byte offset for this slot.
uint64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
@@ -307,8 +295,8 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
// Calculate cachelines needed, clamped to max per instruction.
uint64_t CachelinesNeeded = divideCeil(RemainingBytes, CacheLineSize);
- uint64_t CachelineCount = std::min(
- CachelinesNeeded, static_cast<uint64_t>(MaxCachelinesPerPrefetch));
+ uint64_t CachelineCount =
+ std::min(CachelinesNeeded, MaxCachelinesPerPrefetch);
// The instruction adds 1 to the encoded sdata, so deduct it here.
Res = MCValue::get(static_cast<int64_t>(CachelineCount - 1));
@@ -317,18 +305,13 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
const MCAssembler *Asm) const {
- uint64_t SlotIndex = 0, CodeSizeInBytes = 0, PrefetchBlockOffset = 0;
- if (!evaluateMCExprs(Args, Asm,
- {SlotIndex, CodeSizeInBytes, PrefetchBlockOffset}))
+ uint64_t SlotIndex = 0, CodeSizeInBytes = 0, InstOffset = 0;
+ if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes, InstOffset}))
return false;
- constexpr uint64_t PrefetchInstSize = 8;
-
- uint64_t FirstPrefetch = calcFirstPrefetchOffset(PrefetchBlockOffset);
- uint64_t PrefetchSize =
- calcPrefetchSize(CodeSizeInBytes, PrefetchBlockOffset);
+ int64_t PrefetchSize = calcPrefetchSize(CodeSizeInBytes);
- uint64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
+ int64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
if (SlotOffset >= PrefetchSize) {
// Instruction semantics adds one to sdata when calculating the length of
// the prefetch. This means that even a prefetch instruction with sdata == 0
@@ -339,10 +322,9 @@ bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
return true;
}
// Prefetch is relative to this prefetch instruction's PC.
- uint64_t PC = PrefetchBlockOffset + (SlotIndex * PrefetchInstSize);
- uint64_t Target = FirstPrefetch + SlotOffset;
- uint64_t Offset = Target - PC;
- Res = MCValue::get(static_cast<int64_t>(Offset));
+ int64_t Offset =
+ static_cast<int64_t>(SlotOffset) - static_cast<int64_t>(InstOffset);
+ Res = MCValue::get(Offset);
return true;
}
@@ -456,16 +438,17 @@ AMDGPUMCExpr::createInstPrefSize(const MCExpr *CodeSizeBytes, MCContext &Ctx) {
const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchCachelines(
const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
- const MCExpr *PrefetchBlockOffset, MCContext &Ctx) {
- return create(AGVK_PrefetchCachelines,
- {SlotIndex, CodeSizeBytes, PrefetchBlockOffset}, Ctx);
+ const MCExpr *InstOffset, MCContext &Ctx) {
+ return create(AGVK_PrefetchCachelines, {SlotIndex, CodeSizeBytes, InstOffset},
+ Ctx);
}
-const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchOffset(
- const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
- const MCExpr *PrefetchBlockOffset, MCContext &Ctx) {
- return create(AGVK_PrefetchOffset,
- {SlotIndex, CodeSizeBytes, PrefetchBlockOffset}, Ctx);
+const AMDGPUMCExpr *
+AMDGPUMCExpr::createPrefetchOffset(const MCExpr *SlotIndex,
+ const MCExpr *CodeSizeBytes,
+ const MCExpr *InstOffset, MCContext &Ctx) {
+ return create(AGVK_PrefetchOffset, {SlotIndex, CodeSizeBytes, InstOffset},
+ Ctx);
}
const AMDGPUMCExpr *AMDGPUMCExpr::createLit(LitModifier Lit, int64_t Value,
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index 0db84e172f2ed..bba0e8b01ab3c 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -124,24 +124,25 @@ class AMDGPUMCExpr : public MCTargetExpr {
/// slot.
/// SlotIndex is the 0-based index of the prefetch instruction (0-15).
/// CodeSizeBytes is the total code size in bytes.
- /// PrefetchBlockOffset is the byte offset of the first prefetch instruction
- /// from the function entry.
+ /// InstOffset is the byte offset of this prefetch instruction from the
+ /// function entry.
/// Returns the requested cacheline count minus one, encoded for the 5-bit
/// sdata field.
static const AMDGPUMCExpr *
createPrefetchCachelines(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
- const MCExpr *PrefetchBlockOffset, MCContext &Ctx);
+ const MCExpr *InstOffset, MCContext &Ctx);
/// Create an expression for computing the byte offset for a prefetch slot.
/// SlotIndex is the 0-based index of the prefetch instruction (0-15).
/// CodeSizeBytes is the total code size in bytes.
- /// PrefetchBlockOffset is the byte offset of the first prefetch instruction
- /// from the function entry.
+ /// InstOffset is the byte offset of this prefetch instruction from the
+ /// function entry.
/// Returns the byte offset from the PC of the corresponding prefetch
/// instruction to where this slot should start prefetching.
- static const AMDGPUMCExpr *
- createPrefetchOffset(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
- const MCExpr *PrefetchBlockOffset, MCContext &Ctx);
+ static const AMDGPUMCExpr *createPrefetchOffset(const MCExpr *SlotIndex,
+ const MCExpr *CodeSizeBytes,
+ const MCExpr *InstOffset,
+ MCContext &Ctx);
static const AMDGPUMCExpr *createLit(LitModifier Lit, int64_t Value,
MCContext &Ctx);
diff --git a/llvm/lib/Target/AMDGPU/SIProgramInfo.cpp b/llvm/lib/Target/AMDGPU/SIProgramInfo.cpp
index 713f214bf315a..f0f80f24086c6 100644
--- a/llvm/lib/Target/AMDGPU/SIProgramInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIProgramInfo.cpp
@@ -226,17 +226,26 @@ uint64_t SIProgramInfo::getFunctionCodeSize(const MachineFunction &MF) {
for (const MachineBasicBlock &MBB : MF) {
CodeSize = alignTo(CodeSize, MBB.getAlignment());
+ CodeSize += getMachineBasicBlockCodeSize(MBB, *TII);
+ }
+
+ CodeSizeInBytes = CodeSize;
+ return CodeSize;
+}
+
+uint64_t
+SIProgramInfo::getMachineBasicBlockCodeSize(const MachineBasicBlock &MBB,
+ const SIInstrInfo &TII) {
+ uint64_t CodeSize = 0;
- for (const MachineInstr &MI : MBB) {
- // TODO: CodeSize should account for multiple functions.
+ for (const MachineInstr &MI : MBB) {
+ // TODO: CodeSize should account for multiple functions.
- if (MI.isMetaInstruction())
- continue;
+ if (MI.isMetaInstruction())
+ continue;
- CodeSize += TII->getInstSizeInBytes(MI);
- }
+ CodeSize += TII.getInstSizeInBytes(MI);
}
- CodeSizeInBytes = CodeSize;
return CodeSize;
}
diff --git a/llvm/lib/Target/AMDGPU/SIProgramInfo.h b/llvm/lib/Target/AMDGPU/SIProgramInfo.h
index fb56ebf88c96f..493a332013ed3 100644
--- a/llvm/lib/Target/AMDGPU/SIProgramInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIProgramInfo.h
@@ -26,7 +26,9 @@ namespace llvm {
class GCNSubtarget;
class MCContext;
class MCExpr;
+class MachineBasicBlock;
class MachineFunction;
+class SIInstrInfo;
/// Track resource usage for kernels / entry functions.
struct LLVM_EXTERNAL_VISIBILITY SIProgramInfo {
@@ -107,6 +109,10 @@ struct LLVM_EXTERNAL_VISIBILITY SIProgramInfo {
// Get function code size and cache the value.
uint64_t getFunctionCodeSize(const MachineFunction &MF);
+ // Get machine basic block code size.
+ static uint64_t getMachineBasicBlockCodeSize(const MachineBasicBlock &MBB,
+ const SIInstrInfo &TII);
+
/// Compute the value of the ComputePGMRsrc1 register.
const MCExpr *getComputePGMRSrc1(const GCNSubtarget &ST,
MCContext &Ctx) const;
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 18777c160cc4e..879daea84f3a3 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -41,16 +41,16 @@ define amdgpu_kernel void @below_threshold() {
; Exercise several full 4 KiB prefetch slots and a partial final slot.
; GFX1250-OBJ-LABEL: <partial_final_slot>:
-; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x68, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x1060, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2058, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3050, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4048, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5040, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6038, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7030, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8028, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9020, null, 24
+; GFX1250-OBJ: s_prefetch_inst_pc_rel -0x18, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xfe0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x1fd8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2fd0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3fc8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4fc0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5fb8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6fb0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7fa8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8fa0, null, 25
; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x0, null, 0
define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; GFX1250-LABEL: partial_final_slot:
@@ -58,18 +58,28 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT: .Lpref_block_start0:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(0, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(1, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(2, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(3, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(4, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(5, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(6, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(7, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(8, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(9, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(10, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset_00:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_00-partial_final_slot), null, prefetchcachelines(0, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_00-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset_10:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_10-partial_final_slot), null, prefetchcachelines(1, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_10-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset_20:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_20-partial_final_slot), null, prefetchcachelines(2, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_20-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset_30:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_30-partial_final_slot), null, prefetchcachelines(3, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_30-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset_40:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_40-partial_final_slot), null, prefetchcachelines(4, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_40-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset_50:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_50-partial_final_slot), null, prefetchcachelines(5, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_50-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset_60:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_60-partial_final_slot), null, prefetchcachelines(6, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_60-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset_70:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_70-partial_final_slot), null, prefetchcachelines(7, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_70-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset_80:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_80-partial_final_slot), null, prefetchcachelines(8, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_80-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset_90:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_90-partial_final_slot), null, prefetchcachelines(9, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_90-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset_100:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_100-partial_final_slot), null, prefetchcachelines(10, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_100-partial_final_slot)
; GFX1250-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
; GFX1250-NEXT: v_mov_b32_e32 v0, 0
; GFX1250-NEXT: ;;#ASMSTART
@@ -98,48 +108,62 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
ret void
}
-; Exercise the 16-instruction limit and reserve the cache line containing the
-; first prefetch instruction instead of attempting to replace all 64 KiB.
+; Exercise the 16-instruction limit for the complete 64 KiB ICache range.
; GFX1250-OBJ-LABEL: <cache_size_limit>:
-; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x68, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x1060, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2058, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3050, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4048, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5040, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6038, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7030, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8028, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9020, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xa018, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xb010, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xc008, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xd000, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xdff8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xeff0, null, 30
+; GFX1250-OBJ: s_prefetch_inst_pc_rel -0x18, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xfe0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x1fd8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2fd0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3fc8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4fc0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5fb8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6fb0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7fa8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8fa0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9f98, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xaf90, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xbf88, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xcf80, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xdf78, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xef70, null, 31
define amdgpu_kernel void @cache_size_limit() {
; GFX1250-LABEL: cache_size_limit:
; GFX1250: ; %bb.0:
; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT: .Lpref_block_start1:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(0, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(1, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(2, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(3, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(4, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(5, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(6, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(7, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(8, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(9, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(10, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(11, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(12, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(13, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(14, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(15, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_01:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_01-cache_size_limit), null, prefetchcachelines(0, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_01-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_11:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_11-cache_size_limit), null, prefetchcachelines(1, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_11-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_21:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_21-cache_size_limit), null, prefetchcachelines(2, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_21-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_31:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_31-cache_size_limit), null, prefetchcachelines(3, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_31-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_41:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_41-cache_size_limit), null, prefetchcachelines(4, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_41-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_51:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_51-cache_size_limit), null, prefetchcachelines(5, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_51-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_61:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_61-cache_size_limit), null, prefetchcachelines(6, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_61-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_71:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_71-cache_size_limit), null, prefetchcachelines(7, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_71-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_81:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_81-cache_size_limit), null, prefetchcachelines(8, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_81-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_91:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_91-cache_size_limit), null, prefetchcachelines(9, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_91-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_101:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_101-cache_size_limit), null, prefetchcachelines(10, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_101-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_110:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_110-cache_size_limit), null, prefetchcachelines(11, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_110-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_120:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_120-cache_size_limit), null, prefetchcachelines(12, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_120-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_130:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_130-cache_size_limit), null, prefetchcachelines(13, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_130-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_140:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_140-cache_size_limit), null, prefetchcachelines(14, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_140-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset_150:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_150-cache_size_limit), null, prefetchcachelines(15, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_150-cache_size_limit)
; GFX1250-NEXT: ;;#ASMSTART
; GFX1250-NEXT: .space 65536
; GFX1250-NEXT: ;;#ASMEND
@@ -159,6 +183,125 @@ define amdgpu_kernel void @cache_size_limit() {
ret void
}
+; The shared exit block post-dominates the entry block. Prefetches for code
+; beyond the branch arms are inserted there, rather than all in the entry.
+declare i32 @llvm.amdgcn.workgroup.id.x()
+
+define amdgpu_kernel void @postdominated_prefetch() {
+; GFX1250-LABEL: postdominated_prefetch:
+; GFX1250: ; %bb.0: ; %entry
+; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: .Lpref_inst_offset_02:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_02-postdominated_prefetch), null, prefetchcachelines(0, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_02-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset_12:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_12-postdominated_prefetch), null, prefetchcachelines(1, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_12-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset_22:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_22-postdominated_prefetch), null, prefetchcachelines(2, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_22-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset_32:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_32-postdominated_prefetch), null, prefetchcachelines(3, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_32-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset_42:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch), null, prefetchcachelines(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset_52:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch), null, prefetchcachelines(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch)
+; GFX1250-NEXT: s_bfe_u32 s0, ttmp6, 0x4000c
+; GFX1250-NEXT: s_and_b32 s1, ttmp6, 15
+; GFX1250-NEXT: s_add_co_i32 s0, s0, 1
+; GFX1250-NEXT: s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
+; GFX1250-NEXT: s_mul_i32 s0, ttmp9, s0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_add_co_i32 s1, s1, s0
+; GFX1250-NEXT: s_cmp_eq_u32 s2, 0
+; GFX1250-NEXT: s_cselect_b32 s0, ttmp9, s1
+; GFX1250-NEXT: s_cmp_lg_u32 s0, 0
+; GFX1250-NEXT: s_mov_b32 s0, 0
+; GFX1250-NEXT: s_cbranch_scc0 .LBB3_4
+; GFX1250-NEXT: ; %bb.1: ; %else
+; GFX1250-NEXT: ;;#ASMSTART
+; GFX1250-NEXT: .space 8000
+; GFX1250-NEXT: ;;#ASMEND
+; GFX1250-NEXT: s_and_not1_b32 vcc_lo, exec_lo, s0
+; GFX1250-NEXT: s_cbranch_vccnz .LBB3_3
+; GFX1250-NEXT: .LBB3_2: ; %then
+; GFX1250-NEXT: ;;#ASMSTART
+; GFX1250-NEXT: .space 8000
+; GFX1250-NEXT: ;;#ASMEND
+; GFX1250-NEXT: .LBB3_3: ; %join
+; GFX1250-NEXT: .Lpref_inst_offset_62:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_62-postdominated_prefetch), null, prefetchcachelines(6, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_62-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset_72:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_72-postdominated_prefetch), null, prefetchcachelines(7, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_72-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset_82:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_82-postdominated_prefetch), null, prefetchcachelines(8, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_82-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset_92:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_92-postdominated_prefetch), null, prefetchcachelines(9, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_92-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset_102:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_102-postdominated_prefetch), null, prefetchcachelines(10, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_102-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset_111:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_111-postdominated_prefetch), null, prefetchcachelines(11, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_111-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset_121:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_121-postdominated_prefetch), null, prefetchcachelines(12, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_121-postdominated_prefetch)
+; GFX1250-NEXT: ;;#ASMSTART
+; GFX1250-NEXT: .space 32000
+; GFX1250-NEXT: ;;#ASMEND
+; GFX1250-NEXT: s_endpgm
+; GFX1250-NEXT: .LBB3_4:
+; GFX1250-NEXT: s_branch .LBB3_2
+; GFX1250-NEXT: .Lpref_func_end2:
+;
+; NO-PREFETCH-LABEL: postdominated_prefetch:
+; NO-PREFETCH: ; %bb.0: ; %entry
+; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; NO-PREFETCH-NEXT: v_nop
+; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT: s_bfe_u32 s0, ttmp6, 0x4000c
+; NO-PREFETCH-NEXT: s_and_b32 s1, ttmp6, 15
+; NO-PREFETCH-NEXT: s_add_co_i32 s0, s0, 1
+; NO-PREFETCH-NEXT: s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
+; NO-PREFETCH-NEXT: s_mul_i32 s0, ttmp9, s0
+; NO-PREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; NO-PREFETCH-NEXT: s_add_co_i32 s1, s1, s0
+; NO-PREFETCH-NEXT: s_cmp_eq_u32 s2, 0
+; NO-PREFETCH-NEXT: s_cselect_b32 s0, ttmp9, s1
+; NO-PREFETCH-NEXT: s_cmp_lg_u32 s0, 0
+; NO-PREFETCH-NEXT: s_mov_b32 s0, 0
+; NO-PREFETCH-NEXT: s_cbranch_scc0 .LBB3_4
+; NO-PREFETCH-NEXT: ; %bb.1: ; %else
+; NO-PREFETCH-NEXT: ;;#ASMSTART
+; NO-PREFETCH-NEXT: .space 8000
+; NO-PREFETCH-NEXT: ;;#ASMEND
+; NO-PREFETCH-NEXT: s_and_not1_b32 vcc_lo, exec_lo, s0
+; NO-PREFETCH-NEXT: s_cbranch_vccnz .LBB3_3
+; NO-PREFETCH-NEXT: .LBB3_2: ; %then
+; NO-PREFETCH-NEXT: ;;#ASMSTART
+; NO-PREFETCH-NEXT: .space 8000
+; NO-PREFETCH-NEXT: ;;#ASMEND
+; NO-PREFETCH-NEXT: .LBB3_3: ; %join
+; NO-PREFETCH-NEXT: ;;#ASMSTART
+; NO-PREFETCH-NEXT: .space 32000
+; NO-PREFETCH-NEXT: ;;#ASMEND
+; NO-PREFETCH-NEXT: s_endpgm
+; NO-PREFETCH-NEXT: .LBB3_4:
+; NO-PREFETCH-NEXT: s_branch .LBB3_2
+entry:
+ %id = call i32 @llvm.amdgcn.workgroup.id.x()
+ %cond = icmp eq i32 %id, 0
+ br i1 %cond, label %then, label %else
+
+then:
+ call void asm sideeffect ".space 8000", ""()
+ br label %join
+
+else:
+ call void asm sideeffect ".space 8000", ""()
+ br label %join
+
+join:
+ call void asm sideeffect ".space 32000", ""()
+ ret void
+}
+
; Non-entry functions must not get explicit prefetch instructions regardless
; of their size.
define void @called_function() {
>From 279b50fbfcf752b76acfd1223241f4a96e7f9cb1 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 7 Aug 2026 06:55:07 -0500
Subject: [PATCH 10/22] Add CPop and update tests
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
.../AMDGPU/AMDGPUInsertICachePrefetch.cpp | 3 +-
llvm/test/CodeGen/AMDGPU/icache-prefetch.ll | 50 +++++++++++--------
llvm/test/CodeGen/AMDGPU/llc-pipeline.ll | 4 ++
3 files changed, 36 insertions(+), 21 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index be8703c4c9d6c..b331026feed9c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -215,7 +215,8 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
BuildMI(*CandBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
.addImm(0) // offset (placeholder, fixed up later)
.addReg(AMDGPU::SGPR_NULL) // soffset
- .addImm(Prefetches); // sdata (slot index, fixed up later)
+ .addImm(Prefetches) // sdata (slot index, fixed up later)
+ .addImm(0); // cpol
}
}
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 879daea84f3a3..48f0a31e0dbda 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -183,8 +183,8 @@ define amdgpu_kernel void @cache_size_limit() {
ret void
}
-; The shared exit block post-dominates the entry block. Prefetches for code
-; beyond the branch arms are inserted there, rather than all in the entry.
+; The post-dominator chain distributes prefetches among the entry, Flow, and
+; shared exit blocks, rather than placing them all in the entry.
declare i32 @llvm.amdgcn.workgroup.id.x()
define amdgpu_kernel void @postdominated_prefetch() {
@@ -201,10 +201,6 @@ define amdgpu_kernel void @postdominated_prefetch() {
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_22-postdominated_prefetch), null, prefetchcachelines(2, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_22-postdominated_prefetch)
; GFX1250-NEXT: .Lpref_inst_offset_32:
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_32-postdominated_prefetch), null, prefetchcachelines(3, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_32-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset_42:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch), null, prefetchcachelines(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset_52:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch), null, prefetchcachelines(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch)
; GFX1250-NEXT: s_bfe_u32 s0, ttmp6, 0x4000c
; GFX1250-NEXT: s_and_b32 s1, ttmp6, 15
; GFX1250-NEXT: s_add_co_i32 s0, s0, 1
@@ -216,18 +212,29 @@ define amdgpu_kernel void @postdominated_prefetch() {
; GFX1250-NEXT: s_cselect_b32 s0, ttmp9, s1
; GFX1250-NEXT: s_cmp_lg_u32 s0, 0
; GFX1250-NEXT: s_mov_b32 s0, 0
-; GFX1250-NEXT: s_cbranch_scc0 .LBB3_4
+; GFX1250-NEXT: s_cbranch_scc0 .LBB3_2
; GFX1250-NEXT: ; %bb.1: ; %else
; GFX1250-NEXT: ;;#ASMSTART
; GFX1250-NEXT: .space 8000
; GFX1250-NEXT: ;;#ASMEND
-; GFX1250-NEXT: s_and_not1_b32 vcc_lo, exec_lo, s0
-; GFX1250-NEXT: s_cbranch_vccnz .LBB3_3
-; GFX1250-NEXT: .LBB3_2: ; %then
+; GFX1250-NEXT: s_branch .LBB3_3
+; GFX1250-NEXT: .LBB3_2:
+; GFX1250-NEXT: s_mov_b32 s0, -1
+; GFX1250-NEXT: .LBB3_3: ; %Flow
+; GFX1250-NEXT: .Lpref_inst_offset_42:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch), null, prefetchcachelines(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset_52:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch), null, prefetchcachelines(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_and_b32 s0, s0, exec_lo
+; GFX1250-NEXT: s_cselect_b32 s0, 1, 0
+; GFX1250-NEXT: s_cmp_lg_u32 s0, 1
+; GFX1250-NEXT: s_cbranch_scc1 .LBB3_5
+; GFX1250-NEXT: ; %bb.4: ; %then
; GFX1250-NEXT: ;;#ASMSTART
; GFX1250-NEXT: .space 8000
; GFX1250-NEXT: ;;#ASMEND
-; GFX1250-NEXT: .LBB3_3: ; %join
+; GFX1250-NEXT: .LBB3_5: ; %join
; GFX1250-NEXT: .Lpref_inst_offset_62:
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_62-postdominated_prefetch), null, prefetchcachelines(6, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_62-postdominated_prefetch)
; GFX1250-NEXT: .Lpref_inst_offset_72:
@@ -246,8 +253,6 @@ define amdgpu_kernel void @postdominated_prefetch() {
; GFX1250-NEXT: .space 32000
; GFX1250-NEXT: ;;#ASMEND
; GFX1250-NEXT: s_endpgm
-; GFX1250-NEXT: .LBB3_4:
-; GFX1250-NEXT: s_branch .LBB3_2
; GFX1250-NEXT: .Lpref_func_end2:
;
; NO-PREFETCH-LABEL: postdominated_prefetch:
@@ -266,24 +271,29 @@ define amdgpu_kernel void @postdominated_prefetch() {
; NO-PREFETCH-NEXT: s_cselect_b32 s0, ttmp9, s1
; NO-PREFETCH-NEXT: s_cmp_lg_u32 s0, 0
; NO-PREFETCH-NEXT: s_mov_b32 s0, 0
-; NO-PREFETCH-NEXT: s_cbranch_scc0 .LBB3_4
+; NO-PREFETCH-NEXT: s_cbranch_scc0 .LBB3_2
; NO-PREFETCH-NEXT: ; %bb.1: ; %else
; NO-PREFETCH-NEXT: ;;#ASMSTART
; NO-PREFETCH-NEXT: .space 8000
; NO-PREFETCH-NEXT: ;;#ASMEND
-; NO-PREFETCH-NEXT: s_and_not1_b32 vcc_lo, exec_lo, s0
-; NO-PREFETCH-NEXT: s_cbranch_vccnz .LBB3_3
-; NO-PREFETCH-NEXT: .LBB3_2: ; %then
+; NO-PREFETCH-NEXT: s_branch .LBB3_3
+; NO-PREFETCH-NEXT: .LBB3_2:
+; NO-PREFETCH-NEXT: s_mov_b32 s0, -1
+; NO-PREFETCH-NEXT: .LBB3_3: ; %Flow
+; NO-PREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; NO-PREFETCH-NEXT: s_and_b32 s0, s0, exec_lo
+; NO-PREFETCH-NEXT: s_cselect_b32 s0, 1, 0
+; NO-PREFETCH-NEXT: s_cmp_lg_u32 s0, 1
+; NO-PREFETCH-NEXT: s_cbranch_scc1 .LBB3_5
+; NO-PREFETCH-NEXT: ; %bb.4: ; %then
; NO-PREFETCH-NEXT: ;;#ASMSTART
; NO-PREFETCH-NEXT: .space 8000
; NO-PREFETCH-NEXT: ;;#ASMEND
-; NO-PREFETCH-NEXT: .LBB3_3: ; %join
+; NO-PREFETCH-NEXT: .LBB3_5: ; %join
; NO-PREFETCH-NEXT: ;;#ASMSTART
; NO-PREFETCH-NEXT: .space 32000
; NO-PREFETCH-NEXT: ;;#ASMEND
; NO-PREFETCH-NEXT: s_endpgm
-; NO-PREFETCH-NEXT: .LBB3_4:
-; NO-PREFETCH-NEXT: s_branch .LBB3_2
entry:
%id = call i32 @llvm.amdgcn.workgroup.id.x()
%cond = icmp eq i32 %id, 0
diff --git a/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll b/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
index 8f88d84280b47..9dd47db0287a9 100644
--- a/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
+++ b/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
@@ -455,6 +455,7 @@
; GCN-O1-NEXT: AMDGPU Insert waits for SGPR read hazards
; GCN-O1-NEXT: AMDGPU Lower VGPR Encoding
; GCN-O1-NEXT: AMDGPU Insert Delay ALU
+; GCN-O1-NEXT: MachinePostDominator Tree Construction
; GCN-O1-NEXT: AMDGPU Insert ICache Prefetch
; GCN-O1-NEXT: Branch relaxation pass
; GCN-O1-NEXT: Register Usage Information Collector Pass
@@ -786,6 +787,7 @@
; GCN-O1-OPTS-NEXT: AMDGPU Insert waits for SGPR read hazards
; GCN-O1-OPTS-NEXT: AMDGPU Lower VGPR Encoding
; GCN-O1-OPTS-NEXT: AMDGPU Insert Delay ALU
+; GCN-O1-OPTS-NEXT: MachinePostDominator Tree Construction
; GCN-O1-OPTS-NEXT: AMDGPU Insert ICache Prefetch
; GCN-O1-OPTS-NEXT: Branch relaxation pass
; GCN-O1-OPTS-NEXT: Register Usage Information Collector Pass
@@ -1122,6 +1124,7 @@
; GCN-O2-NEXT: AMDGPU Insert waits for SGPR read hazards
; GCN-O2-NEXT: AMDGPU Lower VGPR Encoding
; GCN-O2-NEXT: AMDGPU Insert Delay ALU
+; GCN-O2-NEXT: MachinePostDominator Tree Construction
; GCN-O2-NEXT: AMDGPU Insert ICache Prefetch
; GCN-O2-NEXT: Branch relaxation pass
; GCN-O2-NEXT: Register Usage Information Collector Pass
@@ -1473,6 +1476,7 @@
; GCN-O3-NEXT: AMDGPU Insert waits for SGPR read hazards
; GCN-O3-NEXT: AMDGPU Lower VGPR Encoding
; GCN-O3-NEXT: AMDGPU Insert Delay ALU
+; GCN-O3-NEXT: MachinePostDominator Tree Construction
; GCN-O3-NEXT: AMDGPU Insert ICache Prefetch
; GCN-O3-NEXT: Branch relaxation pass
; GCN-O3-NEXT: Register Usage Information Collector Pass
>From 8967681f2b1dbd1d3384b5c95e178d58ac0489da Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 7 Aug 2026 09:07:46 -0500
Subject: [PATCH 11/22] Revert remaining changes to SIPreEmitPeephole
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp | 13 ++++++-------
1 file changed, 6 insertions(+), 7 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index e5b1de7bc4ec9..9b67cdd6be6f4 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -23,7 +23,6 @@
#include "llvm/ADT/Statistic.h"
#include "llvm/CodeGen/MachineDominators.h"
#include "llvm/CodeGen/MachineFunctionPass.h"
-#include "llvm/CodeGen/MachineInstrBuilder.h"
#include "llvm/CodeGen/MachineLoopInfo.h"
#include "llvm/CodeGen/TargetSchedule.h"
#include "llvm/Support/BranchProbability.h"
@@ -48,7 +47,6 @@ struct ModeFieldState {
class SIPreEmitPeephole {
private:
- const GCNSubtarget *ST = nullptr;
const SIInstrInfo *TII = nullptr;
const SIRegisterInfo *TRI = nullptr;
MachineLoopInfo *MLI = nullptr;
@@ -193,7 +191,8 @@ bool SIPreEmitPeephole::optimizeVccBranch(MachineInstr &MI) const {
bool Changed = false;
MachineBasicBlock &MBB = *MI.getParent();
- const bool IsWave32 = ST->isWave32();
+ const GCNSubtarget &ST = MBB.getParent()->getSubtarget<GCNSubtarget>();
+ const bool IsWave32 = ST.isWave32();
const unsigned CondReg = TRI->getVCC();
const unsigned ExecReg = IsWave32 ? AMDGPU::EXEC_LO : AMDGPU::EXEC;
const unsigned And = IsWave32 ? AMDGPU::S_AND_B32 : AMDGPU::S_AND_B64;
@@ -867,8 +866,8 @@ llvm::SIPreEmitPeepholePass::run(MachineFunction &MF,
}
bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
- ST = &MF.getSubtarget<GCNSubtarget>();
- TII = ST->getInstrInfo();
+ const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
+ TII = ST.getInstrInfo();
TRI = &TII->getRegisterInfo();
MLI = LoopInfo;
bool Changed = false;
@@ -893,7 +892,7 @@ bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
}
}
- if (!ST->hasVGPRIndexMode())
+ if (!ST.hasVGPRIndexMode())
continue;
MachineInstr *SetGPRMI = nullptr;
@@ -930,7 +929,7 @@ bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
// side effects.
// Perform the extra MF scans only for supported archs
- if (!ST->hasGFX940Insts())
+ if (!ST.hasGFX940Insts())
return Changed;
for (MachineBasicBlock &MBB : MF) {
// Unpack packed instructions overlapped by MFMAs. This allows the
>From 5778fd0e4b36084e92193aaaaba98a6d67b73483 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 7 Aug 2026 09:09:06 -0500
Subject: [PATCH 12/22] Code formatting
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
.../AMDGPU/AMDGPUInsertICachePrefetch.cpp | 20 +++++++++----------
1 file changed, 9 insertions(+), 11 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index b331026feed9c..dfcc9d2206322 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -83,9 +83,9 @@ class AMDGPUInsertICachePrefetchLegacy : public MachineFunctionPass {
} // end anonymous namespace
-static MachineBasicBlock::iterator
-findMBBInsertionPoint(MachineBasicBlock &MBB, const GCNSubtarget &ST,
- bool IsEntryBlock) {
+static MachineBasicBlock::iterator findMBBInsertionPoint(MachineBasicBlock &MBB,
+ const GCNSubtarget &ST,
+ bool IsEntryBlock) {
MachineBasicBlock::iterator InsertPt = MBB.begin();
// Skip past any instructions that must remain at the very beginning:
@@ -99,8 +99,7 @@ findMBBInsertionPoint(MachineBasicBlock &MBB, const GCNSubtarget &ST,
!IsEntryBlock || !ST.hasRequiresInitialUnclausedVmem();
while (InsertPt != MBB.end()) {
if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction() ||
- (IsEntryBlock &&
- InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32)) {
+ (IsEntryBlock && InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32)) {
++InsertPt;
continue;
}
@@ -156,8 +155,7 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
MachineBasicBlock *CandBB = Node->getBlock();
if (!CandBB)
break;
- if (CandBB == &EntryBB ||
- !isLoopFreeEntryPostDominator(*CandBB, EntryBB))
+ if (CandBB == &EntryBB || !isLoopFreeEntryPostDominator(*CandBB, EntryBB))
continue;
Candidates.push_back(CandBB);
}
@@ -202,10 +200,10 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
// - PrefetchSlack: To make sure the last prefetch has the correct number
// of cache lines.
// - BytesPerPrefetch: To account for the latency of the prefetch.
- unsigned PrefetchesBeforeNext = llvm::divideCeil(
- CandidateOffsets.lookup(Candidates[NextCand]) + PrefetchSlack +
- BytesPerPrefetch,
- BytesPerPrefetch);
+ unsigned PrefetchesBeforeNext =
+ llvm::divideCeil(CandidateOffsets.lookup(Candidates[NextCand]) +
+ PrefetchSlack + BytesPerPrefetch,
+ BytesPerPrefetch);
PrefetchBeforeNext = std::min(PrefetchesBeforeNext, PrefetchBeforeNext);
}
MachineBasicBlock *CandBB = Candidates[Cand];
>From 21066b816269857cd02c7398ffbcbaf5bb42d5a5 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 7 Aug 2026 09:26:37 -0500
Subject: [PATCH 13/22] Skip prefetch if basic block sections are used
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp | 7 +++++++
1 file changed, 7 insertions(+)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index dfcc9d2206322..6738dbc0bed74 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -135,6 +135,13 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
if (!MFI->isEntryFunction())
return false;
+ // Basic block sections can be independently placed by the linker, so the
+ // function does not form the contiguous address range assumed by the
+ // prefetch offset and size calculations.
+ if (MF.hasBBSections() ||
+ MF.getTarget().getBBSectionsType() != BasicBlockSection::None)
+ return false;
+
SIProgramInfo PI;
uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
// The kernel descriptor can specify an instruction prefetch size of up to 256
>From 72b3ffa6e3b82836cfa47e1d1a307031f279ab46 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 20 Aug 2026 08:44:34 -0500
Subject: [PATCH 14/22] Make I$ size and prefetch size configurable
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/AMDGPU.td | 26 +++++++++++
.../AMDGPU/AMDGPUInsertICachePrefetch.cpp | 46 +++++++++++++++----
llvm/lib/Target/AMDGPU/GCNSubtarget.h | 11 +++++
.../CodeGen/AMDGPU/icache-prefetch-config.ll | 39 ++++++++++++++++
4 files changed, 113 insertions(+), 9 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index aabc845b3e0a9..c174ef9fa39ec 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -461,6 +461,28 @@ class SubtargetFeatureInstCacheLineSize <int Value> : SubtargetFeature <
def FeatureInstCacheLineSize64 : SubtargetFeatureInstCacheLineSize<64>;
def FeatureInstCacheLineSize128 : SubtargetFeatureInstCacheLineSize<128>;
+class SubtargetFeatureInstCacheSize <int Value> : SubtargetFeature <
+ "instcachesize"#Value,
+ "InstCacheSize",
+ !cast<string>(Value),
+ "Instruction cache size in bytes."
+>;
+
+def FeatureInstCacheSize32768 : SubtargetFeatureInstCacheSize<32768>;
+def FeatureInstCacheSize65536 : SubtargetFeatureInstCacheSize<65536>;
+
+class SubtargetFeaturePreferredInstPrefSize <int Value> : SubtargetFeature <
+ "preferredinstprefsize"#Value,
+ "PreferredInstPrefSize",
+ !cast<string>(Value),
+ "Preferred instruction prefetch size in bytes."
+>;
+
+def FeaturePreferredInstPrefSize16384
+ : SubtargetFeaturePreferredInstPrefSize<16384>;
+def FeaturePreferredInstPrefSize32768
+ : SubtargetFeaturePreferredInstPrefSize<32768>;
+
class SubtargetFeatureDataCacheLineSize <int Value> : SubtargetFeature <
"datacachelinesize"#Value,
"DataCacheLineSize",
@@ -2447,6 +2469,8 @@ def FeatureISAVersion12 : FeatureSet<
FeatureCvtPkNormVOP2Insts,
FeatureCvtPkNormVOP3Insts,
FeatureSmemPrefetchInsts,
+ FeatureInstCacheSize32768,
+ FeaturePreferredInstPrefSize16384,
FeatureNoF16PseudoScalarTransInlineConstants,
FeatureRealTrue16Insts,
FeatureMTBUFInsts,
@@ -2541,6 +2565,8 @@ def FeatureISAVersion12_50_Common : FeatureSet<
FeatureAsyncLoadToLDSInsts,
FeatureAsyncStoreFromLDSInsts,
FeatureSmemPrefetchInsts,
+ FeatureInstCacheSize65536,
+ FeaturePreferredInstPrefSize32768,
FeatureNoF16PseudoScalarTransInlineConstants,
FeatureRealTrue16Insts,
FeatureVMovB64Inst,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index 6738dbc0bed74..e44a1dc55a0f6 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -27,6 +27,7 @@
#include "llvm/CodeGen/MachinePostDominators.h"
#include "llvm/InitializePasses.h"
#include "llvm/Support/CommandLine.h"
+#include "llvm/Support/ErrorHandling.h"
#include "llvm/TargetParser/Triple.h"
using namespace llvm;
@@ -38,6 +39,11 @@ static cl::opt<bool>
cl::desc("Insert ICache prefetch instructions"),
cl::init(true), cl::Hidden);
+static cl::opt<unsigned> ICachePrefetchSize(
+ "amdgpu-icache-prefetch-size",
+ cl::desc("Override the preferred instruction prefetch size in bytes"),
+ cl::init(0), cl::Hidden);
+
namespace {
class AMDGPUInsertICachePrefetch {
@@ -83,6 +89,27 @@ class AMDGPUInsertICachePrefetchLegacy : public MachineFunctionPass {
} // end anonymous namespace
+static uint64_t getPreferredICachePrefetchSize(const GCNSubtarget &ST) {
+ assert(ST.hasInstPrefSize());
+
+ uint64_t PreferredSize = ST.getPreferredInstPrefSize();
+ if (!ICachePrefetchSize.getNumOccurrences())
+ return PreferredSize;
+
+ uint32_t Mask, Shift, Width, CacheLineSize;
+ ST.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
+ uint64_t MaxPrefetchSize = (uint64_t{1} << Width) * CacheLineSize;
+ if (ICachePrefetchSize == 0 ||
+ ICachePrefetchSize % CacheLineSize != 0 ||
+ ICachePrefetchSize > MaxPrefetchSize)
+ report_fatal_error(
+ Twine("-amdgpu-icache-prefetch-size must be a non-zero multiple of ") +
+ Twine(CacheLineSize) + " bytes not exceeding " +
+ Twine(MaxPrefetchSize) + " bytes for " + ST.getCPU());
+
+ return ICachePrefetchSize;
+}
+
static MachineBasicBlock::iterator findMBBInsertionPoint(MachineBasicBlock &MBB,
const GCNSubtarget &ST,
bool IsEntryBlock) {
@@ -123,7 +150,7 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
return false;
const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
- if (!ST.hasICachePrefetch())
+ if (!ST.hasSmemPrefetchInsts() || !ST.hasInstPrefSize())
return false;
// Only run for AMDHSA - this is where kernel descriptors are used and
@@ -142,14 +169,14 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
MF.getTarget().getBBSectionsType() != BasicBlockSection::None)
return false;
+ uint64_t ICacheSize = ST.getInstCacheSize();
+ uint64_t PreferredPrefetchSize = getPreferredICachePrefetchSize(ST);
+ if (ICacheSize == 0 || PreferredPrefetchSize == 0)
+ return false;
+
SIProgramInfo PI;
uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
- // The kernel descriptor can specify an instruction prefetch size of up to 256
- // in INST_PREF_SIZE. At a granularity of 128B, this equals 32KiB of
- // instructions that can be prefetched without inserting explicit prefetch
- // instructions.
- constexpr uint64_t MaxKDPrefetch = 1u << 15;
- if (ProgramSize <= MaxKDPrefetch)
+ if (ProgramSize <= PreferredPrefetchSize)
return false;
const SIInstrInfo *TII = ST.getInstrInfo();
@@ -188,8 +215,9 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
constexpr uint64_t PrefetchSlack = 2 * 1024;
constexpr uint64_t BytesPerPrefetch = 4 * 1024;
// Each prefetch can transfer up to 32 cachelines of 128 bytes = 4KiB.
- // 16 instructions cover 64KiB (the full ICache size).
- constexpr unsigned MaxNumPrefetchInsts = 16;
+ unsigned MaxNumPrefetchInsts = ICacheSize / BytesPerPrefetch;
+ if (MaxNumPrefetchInsts == 0)
+ return false;
unsigned NumPrefetches =
llvm::divideCeil(ProgramSize + PrefetchSlack, BytesPerPrefetch);
NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 28e57bcf6a1c0..647f2a07dc547 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -81,6 +81,11 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
// Instruction cache line size in bytes; set from TableGen subtarget features.
unsigned InstCacheLineSize = 0;
+ // Instruction cache and preferred prefetch sizes in bytes. A zero value
+ // means that the target does not provide the corresponding policy.
+ unsigned InstCacheSize = 0;
+ unsigned PreferredInstPrefSize = 0;
+
// Data (VMEM) cache line size in bytes; set from TableGen subtarget features.
unsigned DataCacheLineSize = 0;
@@ -205,6 +210,12 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
/// Instruction cache line size in bytes (64 for pre-GFX11, 128 for GFX11+).
unsigned getInstCacheLineSize() const { return InstCacheLineSize; }
+ unsigned getInstCacheSize() const { return InstCacheSize; }
+
+ unsigned getPreferredInstPrefSize() const {
+ return PreferredInstPrefSize;
+ }
+
/// Data (VMEM) cache line size in bytes (128 for gfx12), has no use before
/// GFX12.
unsigned getDataCacheLineSize() const { return DataCacheLineSize; }
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
new file mode 100644
index 0000000000000..4a409697481f4
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
@@ -0,0 +1,39 @@
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -o - %s | FileCheck -check-prefix=GFX1200 %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -o - %s | FileCheck -check-prefix=GFX1250 %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-size=16384 -o - %s | FileCheck -check-prefix=OVERRIDE-ENABLE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -amdgpu-icache-prefetch-size=32768 -o - %s | FileCheck -check-prefix=OVERRIDE-DISABLE %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -amdgpu-icache-prefetch-size=32769 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-SIZE %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -amdgpu-icache-prefetch-size=16385 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-SIZE %s
+
+; GFX12 defaults to a 32KiB I-cache and a 16KiB preferred INST_PREF_SIZE.
+; MI450 (gfx1250) defaults to a 64KiB I-cache and a 32KiB preferred size.
+; This 20KiB function is between those two preferred sizes.
+define amdgpu_kernel void @size_between_defaults() {
+; GFX1200-LABEL: size_between_defaults:
+; GFX1200: s_prefetch_inst_pc_rel
+;
+; GFX1250-LABEL: size_between_defaults:
+; GFX1250-NOT: s_prefetch_inst_pc_rel
+; GFX1250: s_endpgm
+;
+; OVERRIDE-ENABLE-LABEL: size_between_defaults:
+; OVERRIDE-ENABLE: s_prefetch_inst_pc_rel
+;
+; OVERRIDE-DISABLE-LABEL: size_between_defaults:
+; OVERRIDE-DISABLE-NOT: s_prefetch_inst_pc_rel
+; OVERRIDE-DISABLE: s_endpgm
+ call void asm sideeffect ".space 20000", ""()
+ ret void
+}
+
+; The GFX12 cache-size feature limits explicit prefetches to eight 4KiB slots.
+define amdgpu_kernel void @gfx12_cache_size_limit() {
+; GFX1200-LABEL: gfx12_cache_size_limit:
+; GFX1200: prefetchoffset(7,
+; GFX1200-NOT: prefetchoffset(8,
+; GFX1200: s_endpgm
+ call void asm sideeffect ".space 65536", ""()
+ ret void
+}
+
+; INVALID-SIZE: LLVM ERROR: -amdgpu-icache-prefetch-size must be a non-zero multiple of 128 bytes not exceeding 32768 bytes for gfx1200
>From e2e85a653d90b8bfb02c8f93bae0a94aea9497cd Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 21 Aug 2026 02:42:32 -0500
Subject: [PATCH 15/22] Combine shader prefetch and prefetch instructions
Combine the initial shader prefetch driven by the INST_PREF_SIZE field
with the insertion of explicit prefetch instructions. The shader
prefetch loads as many bytes as preferred (through option or subtarget
default), the remaining code up to the I$ size limit is loaded with
explicit prefetch instructions inserted into the kernel CFG.
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 11 +-
.../AMDGPU/AMDGPUInsertICachePrefetch.cpp | 43 ++--
llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp | 20 +-
.../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 74 +++----
.../Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h | 17 +-
.../lib/Target/AMDGPU/SIMachineFunctionInfo.h | 15 +-
.../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 8 +
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 4 +
.../CodeGen/AMDGPU/icache-prefetch-config.ll | 13 +-
llvm/test/CodeGen/AMDGPU/icache-prefetch.ll | 209 +++++++-----------
10 files changed, 192 insertions(+), 222 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index fcbc576364e3f..3776390d8deb2 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -264,18 +264,17 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
// right after the function code, so (Lfunc_end - func_sym) gives the
// exact function code size in bytes.
//
- // When ICache prefetch is enabled, we set INST_PREF_SIZE to 1 because
- // the s_prefetch_inst instructions handle the actual prefetching.
+ // When explicit ICache prefetch is enabled, INST_PREF_SIZE prefetches the
+ // initial cache-line prefix and enables scalar prefetch for the first WGP
+ // wave.
if (STM.hasInstPrefSize()) {
uint32_t Mask, Shift, Width, CacheLineSize;
STM.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
const MCExpr *InstPrefSize;
if (MFI.hasICachePrefetch()) {
- // Set INST_PREF_SIZE to 1. This enables prefetch and the first wave on
- // the WGP gets SCALAR_PREFETCH_EN set to 1, while the remaining waves
- // receive 0. This enables the s_prefetch_inst for the first wave.
- InstPrefSize = MCConstantExpr::create(1, Ctx);
+ InstPrefSize =
+ MCConstantExpr::create(MFI.getICachePrefetchLines() - 1, Ctx);
} else {
const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index e44a1dc55a0f6..6823b8fcb632c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -99,8 +99,7 @@ static uint64_t getPreferredICachePrefetchSize(const GCNSubtarget &ST) {
uint32_t Mask, Shift, Width, CacheLineSize;
ST.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
uint64_t MaxPrefetchSize = (uint64_t{1} << Width) * CacheLineSize;
- if (ICachePrefetchSize == 0 ||
- ICachePrefetchSize % CacheLineSize != 0 ||
+ if (ICachePrefetchSize == 0 || ICachePrefetchSize % CacheLineSize != 0 ||
ICachePrefetchSize > MaxPrefetchSize)
report_fatal_error(
Twine("-amdgpu-icache-prefetch-size must be a non-zero multiple of ") +
@@ -174,6 +173,10 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
if (ICacheSize == 0 || PreferredPrefetchSize == 0)
return false;
+ unsigned CacheLineSize = ST.getInstCacheLineSize();
+ unsigned ICacheLines = ICacheSize / CacheLineSize;
+ unsigned DescriptorPrefetchLines = PreferredPrefetchSize / CacheLineSize;
+
SIProgramInfo PI;
uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
if (ProgramSize <= PreferredPrefetchSize)
@@ -215,11 +218,16 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
constexpr uint64_t PrefetchSlack = 2 * 1024;
constexpr uint64_t BytesPerPrefetch = 4 * 1024;
// Each prefetch can transfer up to 32 cachelines of 128 bytes = 4KiB.
- unsigned MaxNumPrefetchInsts = ICacheSize / BytesPerPrefetch;
+ constexpr unsigned CacheLinesPerPrefetch = 32;
+ unsigned MaxNumPrefetchInsts = llvm::divideCeil(
+ ICacheLines - DescriptorPrefetchLines, CacheLinesPerPrefetch);
if (MaxNumPrefetchInsts == 0)
return false;
- unsigned NumPrefetches =
- llvm::divideCeil(ProgramSize + PrefetchSlack, BytesPerPrefetch);
+
+ MFI->setICachePrefetchLines(DescriptorPrefetchLines);
+ uint64_t ProgramPrefetchSize = ProgramSize + PrefetchSlack;
+ unsigned NumPrefetches = llvm::divideCeil(
+ ProgramPrefetchSize - PreferredPrefetchSize, BytesPerPrefetch);
NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
size_t NumCandidates = Candidates.size();
@@ -229,31 +237,36 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
// prefetch latency.
for (size_t Cand = 0, NextCand = 1; Cand < NumCandidates;
++Cand, ++NextCand) {
- unsigned PrefetchBeforeNext = NumPrefetches;
+ unsigned TargetPrefetchCount = NumPrefetches;
if (NextCand < NumCandidates) {
// To the offset of the next candidate we add:
// - PrefetchSlack: To make sure the last prefetch has the correct number
// of cache lines.
// - BytesPerPrefetch: To account for the latency of the prefetch.
- unsigned PrefetchesBeforeNext =
- llvm::divideCeil(CandidateOffsets.lookup(Candidates[NextCand]) +
- PrefetchSlack + BytesPerPrefetch,
- BytesPerPrefetch);
- PrefetchBeforeNext = std::min(PrefetchesBeforeNext, PrefetchBeforeNext);
+ uint64_t CandidatePrefetchSize =
+ CandidateOffsets.lookup(Candidates[NextCand]) + PrefetchSlack +
+ BytesPerPrefetch;
+
+ if (CandidatePrefetchSize <= PreferredPrefetchSize)
+ continue;
+
+ unsigned PrefetchesBeforeNext = llvm::divideCeil(
+ CandidatePrefetchSize - PreferredPrefetchSize, BytesPerPrefetch);
+ TargetPrefetchCount = std::min(PrefetchesBeforeNext, TargetPrefetchCount);
}
MachineBasicBlock *CandBB = Candidates[Cand];
MachineBasicBlock::iterator InsertPt =
findMBBInsertionPoint(*CandBB, ST, CandBB == &EntryBB);
- for (; Prefetches < PrefetchBeforeNext; ++Prefetches) {
+ for (; Prefetches < TargetPrefetchCount; ++Prefetches) {
BuildMI(*CandBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
- .addImm(0) // offset (placeholder, fixed up later)
+ .addImm(DescriptorPrefetchLines + Prefetches * CacheLinesPerPrefetch)
+ // Function-relative target cache-line index, fixed up later.
.addReg(AMDGPU::SGPR_NULL) // soffset
- .addImm(Prefetches) // sdata (slot index, fixed up later)
+ .addImm(0) // sdata (fixed up later)
.addImm(0); // cpol
}
}
- MFI->setHasICachePrefetch(true);
return true;
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
index 7e8f53bee9db6..6941b8f4a3a1d 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
@@ -468,8 +468,9 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
MCInstLowering.lower(MI, TmpInst);
// Fix up S_PREFETCH_INST_PC_REL instructions inserted by the ICache
- // prefetch pass. Replace the slot index in the sdata operand with an
- // MCExpr that computes the cacheline count based on exact code size.
+ // prefetch pass. The provisional offset operand holds a function-relative
+ // target cache-line index; replace it and sdata with expressions that use
+ // final code layout.
if (MI->getOpcode() == AMDGPU::S_PREFETCH_INST_PC_REL) {
const SIMachineFunctionInfo *MFI = MF->getInfo<SIMachineFunctionInfo>();
if (MFI->hasICachePrefetch()) {
@@ -477,12 +478,10 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
constexpr unsigned OffsetIdx = 0;
constexpr unsigned SdataIdx = 2;
- // The sdata operand contains the slot index [0, N) set by the pass.
- int64_t SlotIndex = TmpInst.getOperand(SdataIdx).getImm();
+ int64_t TargetCacheLine = TmpInst.getOperand(OffsetIdx).getImm();
// Emit a symbol for each prefetch instruction to calculate the offset.
- MCSymbol *InstOffsetSym =
- createTempSymbol("pref_inst_offset_" + Twine(SlotIndex));
+ MCSymbol *InstOffsetSym = createTempSymbol("pref_inst_offset");
OutStreamer->emitLabel(InstOffsetSym);
// Create MCExpr for code size using label subtraction.
@@ -491,9 +490,8 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
MCSymbolRefExpr::create(getPrefetchEndSym(), OutContext),
MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
- // Create MCExpr for the slot index.
- const MCExpr *SlotIndexExpr =
- MCConstantExpr::create(SlotIndex, OutContext);
+ const MCExpr *TargetCacheLineExpr =
+ MCConstantExpr::create(TargetCacheLine, OutContext);
// Create an MCExpr for this prefetch instruction's offset in the
// function.
@@ -504,9 +502,9 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
// Create MCExprs that will be evaluated at fixup time when symbol
// positions are known.
const MCExpr *CachelinesExpr = AMDGPUMCExpr::createPrefetchCachelines(
- SlotIndexExpr, CodeSizeExpr, PrefetchInstOffset, OutContext);
+ TargetCacheLineExpr, CodeSizeExpr, PrefetchInstOffset, OutContext);
const MCExpr *OffsetExpr = AMDGPUMCExpr::createPrefetchOffset(
- SlotIndexExpr, CodeSizeExpr, PrefetchInstOffset, OutContext);
+ TargetCacheLineExpr, CodeSizeExpr, PrefetchInstOffset, OutContext);
// Replace the offset and sdata operands with MCExprs.
TmpInst.getOperand(OffsetIdx) = MCOperand::createExpr(OffsetExpr);
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index f0bbab45f6e06..3fe152d831a31 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -253,50 +253,31 @@ bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
return true;
}
-static uint64_t calcPrefetchSize(uint64_t CodeSizeInBytes) {
- constexpr uint64_t MaxPrefetchSize = 64 * 1024; // 64KB ICache
- return std::min(CodeSizeInBytes, MaxPrefetchSize);
-}
-
-static uint64_t calcPrefetchSlotOffset(uint64_t SlotIndex) {
- constexpr unsigned MaxCachelinesPerPrefetch = 32;
- constexpr unsigned CacheLineSize = 128;
- constexpr unsigned BytesPerPrefetch =
- MaxCachelinesPerPrefetch * CacheLineSize;
- return SlotIndex * BytesPerPrefetch;
-}
-
bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
const MCAssembler *Asm) const {
- uint64_t SlotIndex = 0, CodeSizeInBytes = 0, InstOffset = 0;
- if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes, InstOffset}))
+ uint64_t TargetCacheLine = 0, CodeSizeInBytes = 0, InstOffset = 0;
+ if (!evaluateMCExprs(Args, Asm,
+ {TargetCacheLine, CodeSizeInBytes, InstOffset}))
return false;
- // Constants for prefetch calculation.
- // Each instruction can prefetch up to 32 cachelines (5-bit sdata field, plus
- // one added). Cacheline size is 128 bytes. Each slot covers 4KiB.
+ const MCSubtargetInfo *STI = Ctx.getSubtargetInfo();
constexpr uint64_t MaxCachelinesPerPrefetch = 32;
- constexpr unsigned CacheLineSize = 128;
- uint64_t PrefetchSize = calcPrefetchSize(CodeSizeInBytes);
-
- // Calculate the byte offset for this slot.
- uint64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
+ unsigned CacheLineSize = AMDGPU::IsaInfo::getInstCacheLineSize(*STI);
+ uint64_t ICacheLines =
+ AMDGPU::IsaInfo::getInstCacheSize(*STI) / CacheLineSize;
+ uint64_t CodeSizeInLines = divideCeil(CodeSizeInBytes, CacheLineSize);
+ uint64_t PrefetchEnd = std::min(CodeSizeInLines, ICacheLines);
- // If this slot starts beyond the prefetchable region, use the minimum
+ // If this target starts beyond the prefetchable region, use the minimum
// encoded prefetch size. evaluatePrefetchOffset() targets the instruction's
// own cache line for such slots.
- if (SlotOffset >= PrefetchSize) {
+ if (TargetCacheLine >= PrefetchEnd) {
Res = MCValue::get(static_cast<int64_t>(0));
return true;
}
- // Calculate remaining bytes from this slot's offset.
- uint64_t RemainingBytes = PrefetchSize - SlotOffset;
-
- // Calculate cachelines needed, clamped to max per instruction.
- uint64_t CachelinesNeeded = divideCeil(RemainingBytes, CacheLineSize);
uint64_t CachelineCount =
- std::min(CachelinesNeeded, MaxCachelinesPerPrefetch);
+ std::min(PrefetchEnd - TargetCacheLine, MaxCachelinesPerPrefetch);
// The instruction adds 1 to the encoded sdata, so deduct it here.
Res = MCValue::get(static_cast<int64_t>(CachelineCount - 1));
@@ -305,14 +286,17 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
const MCAssembler *Asm) const {
- uint64_t SlotIndex = 0, CodeSizeInBytes = 0, InstOffset = 0;
- if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes, InstOffset}))
+ uint64_t TargetCacheLine = 0, CodeSizeInBytes = 0, InstOffset = 0;
+ if (!evaluateMCExprs(Args, Asm,
+ {TargetCacheLine, CodeSizeInBytes, InstOffset}))
return false;
- int64_t PrefetchSize = calcPrefetchSize(CodeSizeInBytes);
-
- int64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
- if (SlotOffset >= PrefetchSize) {
+ const MCSubtargetInfo *STI = Ctx.getSubtargetInfo();
+ unsigned CacheLineSize = AMDGPU::IsaInfo::getInstCacheLineSize(*STI);
+ uint64_t ICacheLines =
+ AMDGPU::IsaInfo::getInstCacheSize(*STI) / CacheLineSize;
+ uint64_t CodeSizeInLines = divideCeil(CodeSizeInBytes, CacheLineSize);
+ if (TargetCacheLine >= std::min(CodeSizeInLines, ICacheLines)) {
// Instruction semantics adds one to sdata when calculating the length of
// the prefetch. This means that even a prefetch instruction with sdata == 0
// still performs a prefetch. Therefore, to make this prefetch neutral, we
@@ -322,8 +306,8 @@ bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
return true;
}
// Prefetch is relative to this prefetch instruction's PC.
- int64_t Offset =
- static_cast<int64_t>(SlotOffset) - static_cast<int64_t>(InstOffset);
+ int64_t Offset = static_cast<int64_t>(TargetCacheLine * CacheLineSize) -
+ static_cast<int64_t>(InstOffset);
Res = MCValue::get(Offset);
return true;
}
@@ -437,18 +421,18 @@ AMDGPUMCExpr::createInstPrefSize(const MCExpr *CodeSizeBytes, MCContext &Ctx) {
}
const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchCachelines(
- const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
+ const MCExpr *TargetCacheLine, const MCExpr *CodeSizeBytes,
const MCExpr *InstOffset, MCContext &Ctx) {
- return create(AGVK_PrefetchCachelines, {SlotIndex, CodeSizeBytes, InstOffset},
- Ctx);
+ return create(AGVK_PrefetchCachelines,
+ {TargetCacheLine, CodeSizeBytes, InstOffset}, Ctx);
}
const AMDGPUMCExpr *
-AMDGPUMCExpr::createPrefetchOffset(const MCExpr *SlotIndex,
+AMDGPUMCExpr::createPrefetchOffset(const MCExpr *TargetCacheLine,
const MCExpr *CodeSizeBytes,
const MCExpr *InstOffset, MCContext &Ctx) {
- return create(AGVK_PrefetchOffset, {SlotIndex, CodeSizeBytes, InstOffset},
- Ctx);
+ return create(AGVK_PrefetchOffset,
+ {TargetCacheLine, CodeSizeBytes, InstOffset}, Ctx);
}
const AMDGPUMCExpr *AMDGPUMCExpr::createLit(LitModifier Lit, int64_t Value,
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index bba0e8b01ab3c..0f9efa69eacee 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -120,26 +120,27 @@ class AMDGPUMCExpr : public MCTargetExpr {
static const AMDGPUMCExpr *createInstPrefSize(const MCExpr *CodeSizeBytes,
MCContext &Ctx);
- /// Create an expression for computing the encoded sdata field for a prefetch
- /// slot.
- /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
+ /// Create an expression for computing the encoded sdata field for a
+ /// prefetch target. TargetCacheLine is the function-relative target
+ /// cache-line index.
/// CodeSizeBytes is the total code size in bytes.
/// InstOffset is the byte offset of this prefetch instruction from the
/// function entry.
/// Returns the requested cacheline count minus one, encoded for the 5-bit
/// sdata field.
static const AMDGPUMCExpr *
- createPrefetchCachelines(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
+ createPrefetchCachelines(const MCExpr *TargetCacheLine,
+ const MCExpr *CodeSizeBytes,
const MCExpr *InstOffset, MCContext &Ctx);
- /// Create an expression for computing the byte offset for a prefetch slot.
- /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
+ /// Create an expression for computing the byte offset for a prefetch target.
+ /// TargetCacheLine is the function-relative target cache-line index.
/// CodeSizeBytes is the total code size in bytes.
/// InstOffset is the byte offset of this prefetch instruction from the
/// function entry.
/// Returns the byte offset from the PC of the corresponding prefetch
- /// instruction to where this slot should start prefetching.
- static const AMDGPUMCExpr *createPrefetchOffset(const MCExpr *SlotIndex,
+ /// instruction to where this target should start prefetching.
+ static const AMDGPUMCExpr *createPrefetchOffset(const MCExpr *TargetCacheLine,
const MCExpr *CodeSizeBytes,
const MCExpr *InstOffset,
MCContext &Ctx);
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
index 4b85168bdfed6..78240ecd67e37 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
@@ -490,9 +490,10 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
bool HasNonSpillStackObjects = false;
bool IsStackRealigned = false;
- // Set when ICache prefetch instructions have been inserted in the entry
- // block. This tells AsmPrinter to set rsrc3 INST_PREF_SIZE to 0.
- bool HasICachePrefetch = false;
+ // Number of cache lines prefetched through rsrc3 INST_PREF_SIZE when
+ // explicit ICache prefetch instructions are used. A value of zero means the
+ // function does not use the explicit-prefetch scheme.
+ unsigned ICachePrefetchLines = 0;
unsigned NumSpilledSGPRs = 0;
unsigned NumSpilledVGPRs = 0;
@@ -1130,10 +1131,12 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
IsStackRealigned = Realigned;
}
- bool hasICachePrefetch() const { return HasICachePrefetch; }
+ bool hasICachePrefetch() const { return ICachePrefetchLines != 0; }
- void setHasICachePrefetch(bool Prefetch = true) {
- HasICachePrefetch = Prefetch;
+ unsigned getICachePrefetchLines() const { return ICachePrefetchLines; }
+
+ void setICachePrefetchLines(unsigned Lines) {
+ ICachePrefetchLines = Lines;
}
unsigned getNumSpilledSGPRs() const {
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index e45359719d6c9..32e1a1d7c780c 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1107,6 +1107,14 @@ unsigned getInstCacheLineSize(const MCSubtargetInfo &STI) {
return 64;
}
+unsigned getInstCacheSize(const MCSubtargetInfo &STI) {
+ if (STI.getFeatureBits().test(FeatureInstCacheSize65536))
+ return 64 * 1024;
+ if (STI.getFeatureBits().test(FeatureInstCacheSize32768))
+ return 32 * 1024;
+ return 0;
+}
+
unsigned getWavefrontSize(const MCSubtargetInfo &STI) {
if (STI.getFeatureBits().test(FeatureWavefrontSize16))
return 16;
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index bce059a0c18a7..a427856cb691e 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -186,6 +186,10 @@ inline bool targetIDSettingsConflict(TargetIDSetting Lhs, TargetIDSetting Rhs) {
/// \returns Instruction cache line size in bytes for given subtarget \p STI.
unsigned getInstCacheLineSize(const MCSubtargetInfo &STI);
+/// \returns Instruction cache size in bytes for given subtarget \p STI, or
+/// zero if the subtarget does not provide an instruction-cache policy.
+unsigned getInstCacheSize(const MCSubtargetInfo &STI);
+
/// \returns Wavefront size for given subtarget \p STI.
unsigned getWavefrontSize(const MCSubtargetInfo &STI);
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
index 4a409697481f4..2ff4047c3d781 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
@@ -10,14 +10,16 @@
; This 20KiB function is between those two preferred sizes.
define amdgpu_kernel void @size_between_defaults() {
; GFX1200-LABEL: size_between_defaults:
-; GFX1200: s_prefetch_inst_pc_rel
+; GFX1200: prefetchoffset(128,
+; GFX1200: .amdhsa_inst_pref_size 127
;
; GFX1250-LABEL: size_between_defaults:
; GFX1250-NOT: s_prefetch_inst_pc_rel
; GFX1250: s_endpgm
;
; OVERRIDE-ENABLE-LABEL: size_between_defaults:
-; OVERRIDE-ENABLE: s_prefetch_inst_pc_rel
+; OVERRIDE-ENABLE: prefetchoffset(128,
+; OVERRIDE-ENABLE: .amdhsa_inst_pref_size 127
;
; OVERRIDE-DISABLE-LABEL: size_between_defaults:
; OVERRIDE-DISABLE-NOT: s_prefetch_inst_pc_rel
@@ -26,11 +28,12 @@ define amdgpu_kernel void @size_between_defaults() {
ret void
}
-; The GFX12 cache-size feature limits explicit prefetches to eight 4KiB slots.
+; The GFX12 cache-size feature limits explicit prefetches to the four 4KiB
+; slots remaining after the 16KiB descriptor prefetch.
define amdgpu_kernel void @gfx12_cache_size_limit() {
; GFX1200-LABEL: gfx12_cache_size_limit:
-; GFX1200: prefetchoffset(7,
-; GFX1200-NOT: prefetchoffset(8,
+; GFX1200: prefetchoffset(224,
+; GFX1200-NOT: prefetchoffset(256,
; GFX1200: s_endpgm
call void asm sideeffect ".space 65536", ""()
ret void
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 48f0a31e0dbda..7759aa017933d 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -5,6 +5,9 @@
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=false -o - %s | \
; RUN: FileCheck -check-prefix=NO-PREFETCH %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -amdgpu-icache-prefetch-size=16384 -o - %s | \
+; RUN: FileCheck -check-prefix=GFX1250-DIST %s
+
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
; RUN: llvm-objdump -d %t.o | FileCheck -check-prefix=GFX1250-OBJ %s
@@ -18,9 +21,10 @@
define amdgpu_kernel void @below_threshold() {
; GFX1250-LABEL: below_threshold:
; GFX1250: ; %bb.0:
-; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; GFX1250-NEXT: v_nop
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: ;;#ASMSTART
; GFX1250-NEXT: .space 30000
; GFX1250-NEXT: ;;#ASMEND
@@ -28,9 +32,10 @@ define amdgpu_kernel void @below_threshold() {
;
; NO-PREFETCH-LABEL: below_threshold:
; NO-PREFETCH: ; %bb.0:
-; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; NO-PREFETCH-NEXT: v_nop
; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT: s_mov_b64 s[64:65], 0
+; NO-PREFETCH-NEXT: v_nop
+; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; NO-PREFETCH-NEXT: ;;#ASMSTART
; NO-PREFETCH-NEXT: .space 30000
; NO-PREFETCH-NEXT: ;;#ASMEND
@@ -39,47 +44,25 @@ define amdgpu_kernel void @below_threshold() {
ret void
}
-; Exercise several full 4 KiB prefetch slots and a partial final slot.
+; The descriptor prefetches the first 32KiB; explicit requests cover the
+; remaining code in the 64KiB I-cache window.
; GFX1250-OBJ-LABEL: <partial_final_slot>:
-; GFX1250-OBJ: s_prefetch_inst_pc_rel -0x18, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xfe0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x1fd8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2fd0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3fc8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4fc0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5fb8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6fb0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7fa8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8fa0, null, 25
+; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x7ff8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8ff0, null, 25
; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x0, null, 0
define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; GFX1250-LABEL: partial_final_slot:
; GFX1250: ; %bb.0:
-; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; GFX1250-NEXT: v_nop
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT: .Lpref_inst_offset_00:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_00-partial_final_slot), null, prefetchcachelines(0, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_00-partial_final_slot)
-; GFX1250-NEXT: .Lpref_inst_offset_10:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_10-partial_final_slot), null, prefetchcachelines(1, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_10-partial_final_slot)
-; GFX1250-NEXT: .Lpref_inst_offset_20:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_20-partial_final_slot), null, prefetchcachelines(2, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_20-partial_final_slot)
-; GFX1250-NEXT: .Lpref_inst_offset_30:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_30-partial_final_slot), null, prefetchcachelines(3, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_30-partial_final_slot)
-; GFX1250-NEXT: .Lpref_inst_offset_40:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_40-partial_final_slot), null, prefetchcachelines(4, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_40-partial_final_slot)
-; GFX1250-NEXT: .Lpref_inst_offset_50:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_50-partial_final_slot), null, prefetchcachelines(5, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_50-partial_final_slot)
-; GFX1250-NEXT: .Lpref_inst_offset_60:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_60-partial_final_slot), null, prefetchcachelines(6, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_60-partial_final_slot)
-; GFX1250-NEXT: .Lpref_inst_offset_70:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_70-partial_final_slot), null, prefetchcachelines(7, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_70-partial_final_slot)
-; GFX1250-NEXT: .Lpref_inst_offset_80:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_80-partial_final_slot), null, prefetchcachelines(8, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_80-partial_final_slot)
-; GFX1250-NEXT: .Lpref_inst_offset_90:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_90-partial_final_slot), null, prefetchcachelines(9, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_90-partial_final_slot)
-; GFX1250-NEXT: .Lpref_inst_offset_100:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_100-partial_final_slot), null, prefetchcachelines(10, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_100-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset0:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset1:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset2:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot)
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
; GFX1250-NEXT: v_mov_b32_e32 v0, 0
; GFX1250-NEXT: ;;#ASMSTART
@@ -92,9 +75,10 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
;
; NO-PREFETCH-LABEL: partial_final_slot:
; NO-PREFETCH: ; %bb.0:
-; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; NO-PREFETCH-NEXT: v_nop
; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT: s_mov_b64 s[64:65], 0
+; NO-PREFETCH-NEXT: v_nop
+; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; NO-PREFETCH-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
; NO-PREFETCH-NEXT: v_mov_b32_e32 v0, 0
; NO-PREFETCH-NEXT: ;;#ASMSTART
@@ -108,62 +92,40 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
ret void
}
-; Exercise the 16-instruction limit for the complete 64 KiB ICache range.
+; The 32KiB descriptor prefix leaves eight explicit requests for the complete
+; 64KiB I-cache range.
; GFX1250-OBJ-LABEL: <cache_size_limit>:
-; GFX1250-OBJ: s_prefetch_inst_pc_rel -0x18, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xfe0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x1fd8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2fd0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3fc8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4fc0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5fb8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6fb0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7fa8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8fa0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9f98, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xaf90, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xbf88, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xcf80, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xdf78, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xef70, null, 31
+; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x7ff8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8ff0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9fe8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xafe0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xbfd8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xcfd0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xdfc8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xefc0, null, 31
define amdgpu_kernel void @cache_size_limit() {
; GFX1250-LABEL: cache_size_limit:
; GFX1250: ; %bb.0:
-; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; GFX1250-NEXT: v_nop
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT: .Lpref_inst_offset_01:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_01-cache_size_limit), null, prefetchcachelines(0, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_01-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_11:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_11-cache_size_limit), null, prefetchcachelines(1, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_11-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_21:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_21-cache_size_limit), null, prefetchcachelines(2, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_21-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_31:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_31-cache_size_limit), null, prefetchcachelines(3, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_31-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_41:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_41-cache_size_limit), null, prefetchcachelines(4, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_41-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_51:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_51-cache_size_limit), null, prefetchcachelines(5, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_51-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_61:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_61-cache_size_limit), null, prefetchcachelines(6, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_61-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_71:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_71-cache_size_limit), null, prefetchcachelines(7, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_71-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_81:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_81-cache_size_limit), null, prefetchcachelines(8, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_81-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_91:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_91-cache_size_limit), null, prefetchcachelines(9, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_91-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_101:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_101-cache_size_limit), null, prefetchcachelines(10, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_101-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_110:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_110-cache_size_limit), null, prefetchcachelines(11, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_110-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_120:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_120-cache_size_limit), null, prefetchcachelines(12, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_120-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_130:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_130-cache_size_limit), null, prefetchcachelines(13, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_130-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_140:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_140-cache_size_limit), null, prefetchcachelines(14, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_140-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset_150:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_150-cache_size_limit), null, prefetchcachelines(15, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_150-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset3:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset3-cache_size_limit), null, prefetchcachelines(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset3-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset4:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset4-cache_size_limit), null, prefetchcachelines(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset4-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset5:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset5-cache_size_limit), null, prefetchcachelines(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset5-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset6:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset6-cache_size_limit), null, prefetchcachelines(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset6-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset7:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit), null, prefetchcachelines(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset8:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit), null, prefetchcachelines(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset9:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset10:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit)
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: ;;#ASMSTART
; GFX1250-NEXT: .space 65536
; GFX1250-NEXT: ;;#ASMEND
@@ -172,9 +134,10 @@ define amdgpu_kernel void @cache_size_limit() {
;
; NO-PREFETCH-LABEL: cache_size_limit:
; NO-PREFETCH: ; %bb.0:
-; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; NO-PREFETCH-NEXT: v_nop
; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT: s_mov_b64 s[64:65], 0
+; NO-PREFETCH-NEXT: v_nop
+; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; NO-PREFETCH-NEXT: ;;#ASMSTART
; NO-PREFETCH-NEXT: .space 65536
; NO-PREFETCH-NEXT: ;;#ASMEND
@@ -183,24 +146,25 @@ define amdgpu_kernel void @cache_size_limit() {
ret void
}
-; The post-dominator chain distributes prefetches among the entry, Flow, and
-; shared exit blocks, rather than placing them all in the entry.
+; With the default 32KiB descriptor prefix, explicit prefetches are deferred
+; to the shared exit block.
declare i32 @llvm.amdgcn.workgroup.id.x()
define amdgpu_kernel void @postdominated_prefetch() {
+; GFX1250-DIST-LABEL: postdominated_prefetch:
+; GFX1250-DIST-NOT: s_prefetch_inst_pc_rel
+; GFX1250-DIST: .LBB3_3: ; %Flow
+; GFX1250-DIST-NEXT: .Lpref_inst_offset
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128,
+; GFX1250-DIST: .LBB3_5: ; %join
+; GFX1250-DIST-NEXT: .Lpref_inst_offset
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(192,
; GFX1250-LABEL: postdominated_prefetch:
; GFX1250: ; %bb.0: ; %entry
-; GFX1250-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; GFX1250-NEXT: v_nop
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT: .Lpref_inst_offset_02:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_02-postdominated_prefetch), null, prefetchcachelines(0, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_02-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset_12:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_12-postdominated_prefetch), null, prefetchcachelines(1, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_12-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset_22:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_22-postdominated_prefetch), null, prefetchcachelines(2, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_22-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset_32:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_32-postdominated_prefetch), null, prefetchcachelines(3, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_32-postdominated_prefetch)
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: s_bfe_u32 s0, ttmp6, 0x4000c
; GFX1250-NEXT: s_and_b32 s1, ttmp6, 15
; GFX1250-NEXT: s_add_co_i32 s0, s0, 1
@@ -221,10 +185,6 @@ define amdgpu_kernel void @postdominated_prefetch() {
; GFX1250-NEXT: .LBB3_2:
; GFX1250-NEXT: s_mov_b32 s0, -1
; GFX1250-NEXT: .LBB3_3: ; %Flow
-; GFX1250-NEXT: .Lpref_inst_offset_42:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch), null, prefetchcachelines(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset_52:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch), null, prefetchcachelines(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch)
; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_b32 s0, s0, exec_lo
; GFX1250-NEXT: s_cselect_b32 s0, 1, 0
@@ -235,20 +195,16 @@ define amdgpu_kernel void @postdominated_prefetch() {
; GFX1250-NEXT: .space 8000
; GFX1250-NEXT: ;;#ASMEND
; GFX1250-NEXT: .LBB3_5: ; %join
-; GFX1250-NEXT: .Lpref_inst_offset_62:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_62-postdominated_prefetch), null, prefetchcachelines(6, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_62-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset_72:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_72-postdominated_prefetch), null, prefetchcachelines(7, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_72-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset_82:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_82-postdominated_prefetch), null, prefetchcachelines(8, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_82-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset_92:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_92-postdominated_prefetch), null, prefetchcachelines(9, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_92-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset_102:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_102-postdominated_prefetch), null, prefetchcachelines(10, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_102-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset_111:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_111-postdominated_prefetch), null, prefetchcachelines(11, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_111-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset_121:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_121-postdominated_prefetch), null, prefetchcachelines(12, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_121-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset11:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset11-postdominated_prefetch), null, prefetchcachelines(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset11-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset12:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset12-postdominated_prefetch), null, prefetchcachelines(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset12-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset13:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset13-postdominated_prefetch), null, prefetchcachelines(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset13-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset14:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset14-postdominated_prefetch), null, prefetchcachelines(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset14-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset15:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset15-postdominated_prefetch), null, prefetchcachelines(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset15-postdominated_prefetch)
; GFX1250-NEXT: ;;#ASMSTART
; GFX1250-NEXT: .space 32000
; GFX1250-NEXT: ;;#ASMEND
@@ -257,9 +213,10 @@ define amdgpu_kernel void @postdominated_prefetch() {
;
; NO-PREFETCH-LABEL: postdominated_prefetch:
; NO-PREFETCH: ; %bb.0: ; %entry
-; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; NO-PREFETCH-NEXT: v_nop
; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT: s_mov_b64 s[64:65], 0
+; NO-PREFETCH-NEXT: v_nop
+; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; NO-PREFETCH-NEXT: s_bfe_u32 s0, ttmp6, 0x4000c
; NO-PREFETCH-NEXT: s_and_b32 s1, ttmp6, 15
; NO-PREFETCH-NEXT: s_add_co_i32 s0, s0, 1
>From a86eac524f7d13f9f109d80a019c394b02347bfa Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Wed, 30 Sep 2026 08:53:38 -0500
Subject: [PATCH 16/22] Make treshold and initial prefetch size independent
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/AMDGPU.td | 20 +-
.../AMDGPU/AMDGPUInsertICachePrefetch.cpp | 71 +++--
llvm/lib/Target/AMDGPU/GCNSubtarget.h | 8 +-
.../CodeGen/AMDGPU/icache-prefetch-config.ll | 93 ++++--
llvm/test/CodeGen/AMDGPU/icache-prefetch.ll | 288 ++++++++++++++----
5 files changed, 367 insertions(+), 113 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index c174ef9fa39ec..9e9cdd10d99cb 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -471,17 +471,17 @@ class SubtargetFeatureInstCacheSize <int Value> : SubtargetFeature <
def FeatureInstCacheSize32768 : SubtargetFeatureInstCacheSize<32768>;
def FeatureInstCacheSize65536 : SubtargetFeatureInstCacheSize<65536>;
-class SubtargetFeaturePreferredInstPrefSize <int Value> : SubtargetFeature <
- "preferredinstprefsize"#Value,
- "PreferredInstPrefSize",
+class SubtargetFeatureInitialInstPrefSize <int Value> : SubtargetFeature <
+ "initialinstprefsize"#Value,
+ "InitialInstPrefSize",
!cast<string>(Value),
- "Preferred instruction prefetch size in bytes."
+ "Initial instruction prefetch size in bytes."
>;
-def FeaturePreferredInstPrefSize16384
- : SubtargetFeaturePreferredInstPrefSize<16384>;
-def FeaturePreferredInstPrefSize32768
- : SubtargetFeaturePreferredInstPrefSize<32768>;
+def FeatureInitialInstPrefSize8192
+ : SubtargetFeatureInitialInstPrefSize<8192>;
+def FeatureInitialInstPrefSize16384
+ : SubtargetFeatureInitialInstPrefSize<16384>;
class SubtargetFeatureDataCacheLineSize <int Value> : SubtargetFeature <
"datacachelinesize"#Value,
@@ -2470,7 +2470,7 @@ def FeatureISAVersion12 : FeatureSet<
FeatureCvtPkNormVOP3Insts,
FeatureSmemPrefetchInsts,
FeatureInstCacheSize32768,
- FeaturePreferredInstPrefSize16384,
+ FeatureInitialInstPrefSize16384,
FeatureNoF16PseudoScalarTransInlineConstants,
FeatureRealTrue16Insts,
FeatureMTBUFInsts,
@@ -2566,7 +2566,7 @@ def FeatureISAVersion12_50_Common : FeatureSet<
FeatureAsyncStoreFromLDSInsts,
FeatureSmemPrefetchInsts,
FeatureInstCacheSize65536,
- FeaturePreferredInstPrefSize32768,
+ FeatureInitialInstPrefSize8192,
FeatureNoF16PseudoScalarTransInlineConstants,
FeatureRealTrue16Insts,
FeatureVMovB64Inst,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index 6823b8fcb632c..73766a35e2ed2 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -39,9 +39,14 @@ static cl::opt<bool>
cl::desc("Insert ICache prefetch instructions"),
cl::init(true), cl::Hidden);
-static cl::opt<unsigned> ICachePrefetchSize(
- "amdgpu-icache-prefetch-size",
- cl::desc("Override the preferred instruction prefetch size in bytes"),
+static cl::opt<unsigned> ICachePrefetchInitialSize(
+ "amdgpu-icache-prefetch-initial-size",
+ cl::desc("Override the initial instruction prefetch size in bytes"),
+ cl::init(0), cl::Hidden);
+
+static cl::opt<unsigned> ICachePrefetchThreshold(
+ "amdgpu-icache-prefetch-threshold",
+ cl::desc("Override the explicit instruction prefetch threshold in bytes"),
cl::init(0), cl::Hidden);
namespace {
@@ -89,24 +94,48 @@ class AMDGPUInsertICachePrefetchLegacy : public MachineFunctionPass {
} // end anonymous namespace
-static uint64_t getPreferredICachePrefetchSize(const GCNSubtarget &ST) {
- assert(ST.hasInstPrefSize());
+struct ICachePrefetchConfig {
+ uint64_t InitialSize;
+ uint64_t Threshold;
+};
- uint64_t PreferredSize = ST.getPreferredInstPrefSize();
- if (!ICachePrefetchSize.getNumOccurrences())
- return PreferredSize;
+static ICachePrefetchConfig getICachePrefetchConfig(const GCNSubtarget &ST) {
+ assert(ST.hasInstPrefSize());
uint32_t Mask, Shift, Width, CacheLineSize;
ST.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
- uint64_t MaxPrefetchSize = (uint64_t{1} << Width) * CacheLineSize;
- if (ICachePrefetchSize == 0 || ICachePrefetchSize % CacheLineSize != 0 ||
- ICachePrefetchSize > MaxPrefetchSize)
+ uint64_t DescriptorPrefetchCapacity = (uint64_t{1} << Width) * CacheLineSize;
+ uint64_t InitialSize = ICachePrefetchInitialSize.getNumOccurrences()
+ ? ICachePrefetchInitialSize
+ : ST.getInitialInstPrefSize();
+ uint64_t Threshold = ICachePrefetchThreshold.getNumOccurrences()
+ ? ICachePrefetchThreshold
+ : DescriptorPrefetchCapacity;
+
+ if (InitialSize == 0 || InitialSize % CacheLineSize != 0 ||
+ InitialSize > DescriptorPrefetchCapacity)
report_fatal_error(
- Twine("-amdgpu-icache-prefetch-size must be a non-zero multiple of ") +
+ Twine("-amdgpu-icache-prefetch-initial-size must be a non-zero "
+ "multiple of ") +
Twine(CacheLineSize) + " bytes not exceeding " +
- Twine(MaxPrefetchSize) + " bytes for " + ST.getCPU());
+ Twine(DescriptorPrefetchCapacity) + " bytes for " + ST.getCPU());
+
+ uint64_t ICacheSize = ST.getInstCacheSize();
+ if (Threshold == 0 || Threshold % CacheLineSize != 0 ||
+ Threshold > ICacheSize)
+ report_fatal_error(
+ Twine("-amdgpu-icache-prefetch-threshold must be a non-zero multiple "
+ "of ") +
+ Twine(CacheLineSize) + " bytes not exceeding " + Twine(ICacheSize) +
+ " bytes for " + ST.getCPU());
+
+ if (InitialSize > Threshold)
+ report_fatal_error(
+ Twine("-amdgpu-icache-prefetch-initial-size must not exceed "
+ "-amdgpu-icache-prefetch-threshold for ") +
+ ST.getCPU());
- return ICachePrefetchSize;
+ return {InitialSize, Threshold};
}
static MachineBasicBlock::iterator findMBBInsertionPoint(MachineBasicBlock &MBB,
@@ -169,17 +198,17 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
return false;
uint64_t ICacheSize = ST.getInstCacheSize();
- uint64_t PreferredPrefetchSize = getPreferredICachePrefetchSize(ST);
- if (ICacheSize == 0 || PreferredPrefetchSize == 0)
+ if (ICacheSize == 0)
return false;
+ const ICachePrefetchConfig Config = getICachePrefetchConfig(ST);
unsigned CacheLineSize = ST.getInstCacheLineSize();
unsigned ICacheLines = ICacheSize / CacheLineSize;
- unsigned DescriptorPrefetchLines = PreferredPrefetchSize / CacheLineSize;
+ unsigned DescriptorPrefetchLines = Config.InitialSize / CacheLineSize;
SIProgramInfo PI;
uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
- if (ProgramSize <= PreferredPrefetchSize)
+ if (ProgramSize <= Config.Threshold)
return false;
const SIInstrInfo *TII = ST.getInstrInfo();
@@ -227,7 +256,7 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
MFI->setICachePrefetchLines(DescriptorPrefetchLines);
uint64_t ProgramPrefetchSize = ProgramSize + PrefetchSlack;
unsigned NumPrefetches = llvm::divideCeil(
- ProgramPrefetchSize - PreferredPrefetchSize, BytesPerPrefetch);
+ ProgramPrefetchSize - Config.InitialSize, BytesPerPrefetch);
NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
size_t NumCandidates = Candidates.size();
@@ -247,11 +276,11 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
CandidateOffsets.lookup(Candidates[NextCand]) + PrefetchSlack +
BytesPerPrefetch;
- if (CandidatePrefetchSize <= PreferredPrefetchSize)
+ if (CandidatePrefetchSize <= Config.InitialSize)
continue;
unsigned PrefetchesBeforeNext = llvm::divideCeil(
- CandidatePrefetchSize - PreferredPrefetchSize, BytesPerPrefetch);
+ CandidatePrefetchSize - Config.InitialSize, BytesPerPrefetch);
TargetPrefetchCount = std::min(PrefetchesBeforeNext, TargetPrefetchCount);
}
MachineBasicBlock *CandBB = Candidates[Cand];
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 647f2a07dc547..df1f651de13f0 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -81,10 +81,10 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
// Instruction cache line size in bytes; set from TableGen subtarget features.
unsigned InstCacheLineSize = 0;
- // Instruction cache and preferred prefetch sizes in bytes. A zero value
+ // Instruction cache and initial prefetch sizes in bytes. A zero value
// means that the target does not provide the corresponding policy.
unsigned InstCacheSize = 0;
- unsigned PreferredInstPrefSize = 0;
+ unsigned InitialInstPrefSize = 0;
// Data (VMEM) cache line size in bytes; set from TableGen subtarget features.
unsigned DataCacheLineSize = 0;
@@ -212,9 +212,7 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
unsigned getInstCacheSize() const { return InstCacheSize; }
- unsigned getPreferredInstPrefSize() const {
- return PreferredInstPrefSize;
- }
+ unsigned getInitialInstPrefSize() const { return InitialInstPrefSize; }
/// Data (VMEM) cache line size in bytes (128 for gfx12), has no use before
/// GFX12.
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
index 2ff4047c3d781..574ca53a26363 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
@@ -1,42 +1,83 @@
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -o - %s | FileCheck -check-prefix=GFX1200 %s
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -o - %s | FileCheck -check-prefix=GFX1250 %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-size=16384 -o - %s | FileCheck -check-prefix=OVERRIDE-ENABLE %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -amdgpu-icache-prefetch-size=32768 -o - %s | FileCheck -check-prefix=OVERRIDE-DISABLE %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -amdgpu-icache-prefetch-size=32769 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-SIZE %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -amdgpu-icache-prefetch-size=16385 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-SIZE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -o - %s | FileCheck -check-prefix=INITIAL-16K %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=49152 -o - %s | FileCheck -check-prefix=THRESHOLD-48K %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=THRESHOLD-16K %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=EQUAL-16K %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=0 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=8193 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=32896 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=0 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=32769 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=65664 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=8192 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-ORDER %s
-; GFX12 defaults to a 32KiB I-cache and a 16KiB preferred INST_PREF_SIZE.
-; MI450 (gfx1250) defaults to a 64KiB I-cache and a 32KiB preferred size.
-; This 20KiB function is between those two preferred sizes.
-define amdgpu_kernel void @size_between_defaults() {
-; GFX1200-LABEL: size_between_defaults:
-; GFX1200: prefetchoffset(128,
+; The default threshold is the 32KiB descriptor capacity on both targets.
+; gfx1200 initially prefetches 16KiB, while gfx1250 initially prefetches 8KiB.
+define amdgpu_kernel void @above_default_threshold() {
+; GFX1200-LABEL: above_default_threshold:
+; GFX1200: s_prefetch_inst_pc_rel prefetchoffset(128,
; GFX1200: .amdhsa_inst_pref_size 127
;
-; GFX1250-LABEL: size_between_defaults:
+; GFX1250-LABEL: above_default_threshold:
+; GFX1250: s_prefetch_inst_pc_rel prefetchoffset(64,
+; GFX1250: .amdhsa_inst_pref_size 63
+;
+; INITIAL-16K-LABEL: above_default_threshold:
+; INITIAL-16K: s_prefetch_inst_pc_rel prefetchoffset(128,
+; INITIAL-16K: .amdhsa_inst_pref_size 127
+;
+; THRESHOLD-48K-LABEL: above_default_threshold:
+; THRESHOLD-48K-NOT: s_prefetch_inst_pc_rel
+; THRESHOLD-48K: s_endpgm
+; THRESHOLD-48K: .amdhsa_inst_pref_size ((instprefsize(
+ call void asm sideeffect ".space 40000", ""()
+ ret void
+}
+
+; The gfx1250 prologue and s_endpgm add 32 bytes, making these functions
+; exactly 32KiB and four bytes over 32KiB respectively.
+define amdgpu_kernel void @at_default_threshold() {
+; GFX1250-LABEL: at_default_threshold:
; GFX1250-NOT: s_prefetch_inst_pc_rel
; GFX1250: s_endpgm
+ call void asm sideeffect ".space 32736", ""()
+ ret void
+}
+
+define amdgpu_kernel void @above_default_threshold_by_four() {
+; GFX1250-LABEL: above_default_threshold_by_four:
+; GFX1250: s_prefetch_inst_pc_rel prefetchoffset(64,
+ call void asm sideeffect ".space 32740", ""()
+ ret void
+}
+
+; Lowering only the threshold activates explicit prefetching while retaining
+; the default 8KiB initial descriptor coverage. Setting initial == threshold
+; is also valid and changes the descriptor coverage independently.
+define amdgpu_kernel void @between_override_thresholds() {
+; THRESHOLD-16K-LABEL: between_override_thresholds:
+; THRESHOLD-16K: s_prefetch_inst_pc_rel prefetchoffset(64,
+; THRESHOLD-16K: .amdhsa_inst_pref_size 63
;
-; OVERRIDE-ENABLE-LABEL: size_between_defaults:
-; OVERRIDE-ENABLE: prefetchoffset(128,
-; OVERRIDE-ENABLE: .amdhsa_inst_pref_size 127
-;
-; OVERRIDE-DISABLE-LABEL: size_between_defaults:
-; OVERRIDE-DISABLE-NOT: s_prefetch_inst_pc_rel
-; OVERRIDE-DISABLE: s_endpgm
+; EQUAL-16K-LABEL: between_override_thresholds:
+; EQUAL-16K: s_prefetch_inst_pc_rel prefetchoffset(128,
+; EQUAL-16K: .amdhsa_inst_pref_size 127
call void asm sideeffect ".space 20000", ""()
ret void
}
-; The GFX12 cache-size feature limits explicit prefetches to the four 4KiB
-; slots remaining after the 16KiB descriptor prefetch.
-define amdgpu_kernel void @gfx12_cache_size_limit() {
-; GFX1200-LABEL: gfx12_cache_size_limit:
-; GFX1200: prefetchoffset(224,
-; GFX1200-NOT: prefetchoffset(256,
-; GFX1200: s_endpgm
+; INITIAL-16K-LABEL: cache_size_limit:
+; INITIAL-16K-COUNT-12: s_prefetch_inst_pc_rel
+; INITIAL-16K-NEXT: s_mov_b64
+; GFX1250-LABEL: cache_size_limit:
+; GFX1250-COUNT-14: s_prefetch_inst_pc_rel
+; GFX1250-NEXT: s_mov_b64
+define amdgpu_kernel void @cache_size_limit() {
call void asm sideeffect ".space 65536", ""()
ret void
}
-; INVALID-SIZE: LLVM ERROR: -amdgpu-icache-prefetch-size must be a non-zero multiple of 128 bytes not exceeding 32768 bytes for gfx1200
+; INVALID-INITIAL: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must be a non-zero multiple of 128 bytes not exceeding 32768 bytes for gfx1250
+; INVALID-THRESHOLD: LLVM ERROR: -amdgpu-icache-prefetch-threshold must be a non-zero multiple of 128 bytes not exceeding 65536 bytes for gfx1250
+; INVALID-ORDER: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must not exceed -amdgpu-icache-prefetch-threshold for gfx1250
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 7759aa017933d..286cfb8cb5949 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -5,7 +5,7 @@
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=false -o - %s | \
; RUN: FileCheck -check-prefix=NO-PREFETCH %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -amdgpu-icache-prefetch-size=16384 -o - %s | \
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -amdgpu-icache-prefetch-initial-size=16384 -o - %s | \
; RUN: FileCheck -check-prefix=GFX1250-DIST %s
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
@@ -14,7 +14,7 @@
; Use .space to make the estimated MachineFunction size and final assembled
; size large without spelling out thousands of instructions.
-; A function below the 32 KiB kernel-descriptor prefetch limit must not use
+; A function below the 32 KiB explicit-prefetch threshold must not use
; explicit prefetch instructions.
; GFX1250-OBJ-LABEL: <below_threshold>:
; GFX1250-OBJ-NOT: s_prefetch_inst_pc_rel
@@ -40,26 +40,55 @@ define amdgpu_kernel void @below_threshold() {
; NO-PREFETCH-NEXT: .space 30000
; NO-PREFETCH-NEXT: ;;#ASMEND
; NO-PREFETCH-NEXT: s_endpgm
+;
+; GFX1250-DIST-LABEL: below_threshold:
+; GFX1250-DIST: ; %bb.0:
+; GFX1250-DIST-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-DIST-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-DIST-NEXT: v_nop
+; GFX1250-DIST-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-DIST-NEXT: ;;#ASMSTART
+; GFX1250-DIST-NEXT: .space 30000
+; GFX1250-DIST-NEXT: ;;#ASMEND
+; GFX1250-DIST-NEXT: s_endpgm
call void asm sideeffect ".space 30000", ""()
ret void
}
-; The descriptor prefetches the first 32KiB; explicit requests cover the
+; The descriptor prefetches the first 8KiB; explicit requests cover the
; remaining code in the 64KiB I-cache window.
; GFX1250-OBJ-LABEL: <partial_final_slot>:
-; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x7ff8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8ff0, null, 25
+; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x1ff8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2ff0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3fe8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4fe0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5fd8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6fd0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7fc8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8fc0, null, 25
; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x0, null, 0
define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; GFX1250-LABEL: partial_final_slot:
; GFX1250: ; %bb.0:
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
; GFX1250-NEXT: .Lpref_inst_offset0:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(64, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(64, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
; GFX1250-NEXT: .Lpref_inst_offset1:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(96, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot), null, prefetchcachelines(96, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot)
; GFX1250-NEXT: .Lpref_inst_offset2:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot), null, prefetchcachelines(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset3:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset3-partial_final_slot), null, prefetchcachelines(160, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset3-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset4:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset4-partial_final_slot), null, prefetchcachelines(192, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset4-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset5:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot), null, prefetchcachelines(224, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset6:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot), null, prefetchcachelines(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset7:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset7-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset7-partial_final_slot)
+; GFX1250-NEXT: .Lpref_inst_offset8:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset8-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset8-partial_final_slot)
; GFX1250-NEXT: s_mov_b64 s[64:65], 0
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
@@ -87,42 +116,90 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; NO-PREFETCH-NEXT: s_wait_kmcnt 0x0
; NO-PREFETCH-NEXT: global_store_b32 v0, v0, s[0:1]
; NO-PREFETCH-NEXT: s_endpgm
+;
+; GFX1250-DIST-LABEL: partial_final_slot:
+; GFX1250-DIST: ; %bb.0:
+; GFX1250-DIST-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-DIST-NEXT: .Lpref_inst_offset0:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset1:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot), null, prefetchcachelines(160, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset2:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot), null, prefetchcachelines(192, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset3:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset3-partial_final_slot), null, prefetchcachelines(224, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset3-partial_final_slot)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset4:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset4-partial_final_slot), null, prefetchcachelines(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset4-partial_final_slot)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset5:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset6:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot)
+; GFX1250-DIST-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-DIST-NEXT: v_nop
+; GFX1250-DIST-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-DIST-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
+; GFX1250-DIST-NEXT: v_mov_b32_e32 v0, 0
+; GFX1250-DIST-NEXT: ;;#ASMSTART
+; GFX1250-DIST-NEXT: .space 40000
+; GFX1250-DIST-NEXT: ;;#ASMEND
+; GFX1250-DIST-NEXT: s_wait_kmcnt 0x0
+; GFX1250-DIST-NEXT: global_store_b32 v0, v0, s[0:1]
+; GFX1250-DIST-NEXT: s_endpgm
+; GFX1250-DIST-NEXT: .Lpref_func_end0:
call void asm sideeffect ".space 40000", ""()
store i32 0, ptr addrspace(1) %out
ret void
}
-; The 32KiB descriptor prefix leaves eight explicit requests for the complete
+; The 8KiB descriptor prefix leaves fourteen explicit requests for the complete
; 64KiB I-cache range.
; GFX1250-OBJ-LABEL: <cache_size_limit>:
-; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x7ff8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8ff0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9fe8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xafe0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xbfd8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xcfd0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xdfc8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xefc0, null, 31
+; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x1ff8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2ff0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3fe8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4fe0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5fd8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6fd0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7fc8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8fc0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9fb8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xafb0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xbfa8, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xcfa0, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xdf98, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xef90, null, 31
define amdgpu_kernel void @cache_size_limit() {
; GFX1250-LABEL: cache_size_limit:
; GFX1250: ; %bb.0:
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT: .Lpref_inst_offset3:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset3-cache_size_limit), null, prefetchcachelines(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset3-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset4:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset4-cache_size_limit), null, prefetchcachelines(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset4-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset5:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset5-cache_size_limit), null, prefetchcachelines(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset5-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset6:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset6-cache_size_limit), null, prefetchcachelines(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset6-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset7:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit), null, prefetchcachelines(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit)
-; GFX1250-NEXT: .Lpref_inst_offset8:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit), null, prefetchcachelines(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit)
; GFX1250-NEXT: .Lpref_inst_offset9:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(64, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit), null, prefetchcachelines(64, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit)
; GFX1250-NEXT: .Lpref_inst_offset10:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit)
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(96, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit), null, prefetchcachelines(96, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset11:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset11-cache_size_limit), null, prefetchcachelines(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset11-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset12:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset12-cache_size_limit), null, prefetchcachelines(160, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset12-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset13:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset13-cache_size_limit), null, prefetchcachelines(192, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset13-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset14:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset14-cache_size_limit), null, prefetchcachelines(224, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset14-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset15:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset15-cache_size_limit), null, prefetchcachelines(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset15-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset16:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset16-cache_size_limit), null, prefetchcachelines(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset16-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset17:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit), null, prefetchcachelines(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset18:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit), null, prefetchcachelines(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset19:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset19-cache_size_limit), null, prefetchcachelines(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset19-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset20:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset20-cache_size_limit), null, prefetchcachelines(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset20-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset21:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset21-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset21-cache_size_limit)
+; GFX1250-NEXT: .Lpref_inst_offset22:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset22-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset22-cache_size_limit)
; GFX1250-NEXT: s_mov_b64 s[64:65], 0
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
@@ -142,26 +219,58 @@ define amdgpu_kernel void @cache_size_limit() {
; NO-PREFETCH-NEXT: .space 65536
; NO-PREFETCH-NEXT: ;;#ASMEND
; NO-PREFETCH-NEXT: s_endpgm
+;
+; GFX1250-DIST-LABEL: cache_size_limit:
+; GFX1250-DIST: ; %bb.0:
+; GFX1250-DIST-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-DIST-NEXT: .Lpref_inst_offset7:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit), null, prefetchcachelines(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset8:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit), null, prefetchcachelines(160, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset9:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit), null, prefetchcachelines(192, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset10:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit), null, prefetchcachelines(224, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset11:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset11-cache_size_limit), null, prefetchcachelines(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset11-cache_size_limit)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset12:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset12-cache_size_limit), null, prefetchcachelines(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset12-cache_size_limit)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset13:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset13-cache_size_limit), null, prefetchcachelines(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset13-cache_size_limit)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset14:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset14-cache_size_limit), null, prefetchcachelines(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset14-cache_size_limit)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset15:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset15-cache_size_limit), null, prefetchcachelines(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset15-cache_size_limit)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset16:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset16-cache_size_limit), null, prefetchcachelines(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset16-cache_size_limit)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset17:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset18:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit)
+; GFX1250-DIST-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-DIST-NEXT: v_nop
+; GFX1250-DIST-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-DIST-NEXT: ;;#ASMSTART
+; GFX1250-DIST-NEXT: .space 65536
+; GFX1250-DIST-NEXT: ;;#ASMEND
+; GFX1250-DIST-NEXT: s_endpgm
+; GFX1250-DIST-NEXT: .Lpref_func_end1:
call void asm sideeffect ".space 65536", ""()
ret void
}
-; With the default 32KiB descriptor prefix, explicit prefetches are deferred
-; to the shared exit block.
+; With a 16KiB descriptor prefix override, explicit prefetches are deferred to
+; the shared exit block.
declare i32 @llvm.amdgcn.workgroup.id.x()
define amdgpu_kernel void @postdominated_prefetch() {
-; GFX1250-DIST-LABEL: postdominated_prefetch:
-; GFX1250-DIST-NOT: s_prefetch_inst_pc_rel
-; GFX1250-DIST: .LBB3_3: ; %Flow
-; GFX1250-DIST-NEXT: .Lpref_inst_offset
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128,
-; GFX1250-DIST: .LBB3_5: ; %join
-; GFX1250-DIST-NEXT: .Lpref_inst_offset
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(192,
; GFX1250-LABEL: postdominated_prefetch:
; GFX1250: ; %bb.0: ; %entry
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: .Lpref_inst_offset23:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(64, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch), null, prefetchcachelines(64, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset24:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(96, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch), null, prefetchcachelines(96, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch)
; GFX1250-NEXT: s_mov_b64 s[64:65], 0
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
@@ -185,6 +294,10 @@ define amdgpu_kernel void @postdominated_prefetch() {
; GFX1250-NEXT: .LBB3_2:
; GFX1250-NEXT: s_mov_b32 s0, -1
; GFX1250-NEXT: .LBB3_3: ; %Flow
+; GFX1250-NEXT: .Lpref_inst_offset25:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset25-postdominated_prefetch), null, prefetchcachelines(128, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset25-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset26:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset26-postdominated_prefetch), null, prefetchcachelines(160, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset26-postdominated_prefetch)
; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_b32 s0, s0, exec_lo
; GFX1250-NEXT: s_cselect_b32 s0, 1, 0
@@ -195,16 +308,20 @@ define amdgpu_kernel void @postdominated_prefetch() {
; GFX1250-NEXT: .space 8000
; GFX1250-NEXT: ;;#ASMEND
; GFX1250-NEXT: .LBB3_5: ; %join
-; GFX1250-NEXT: .Lpref_inst_offset11:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset11-postdominated_prefetch), null, prefetchcachelines(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset11-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset12:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset12-postdominated_prefetch), null, prefetchcachelines(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset12-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset13:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset13-postdominated_prefetch), null, prefetchcachelines(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset13-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset14:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset14-postdominated_prefetch), null, prefetchcachelines(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset14-postdominated_prefetch)
-; GFX1250-NEXT: .Lpref_inst_offset15:
-; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset15-postdominated_prefetch), null, prefetchcachelines(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset15-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset27:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset27-postdominated_prefetch), null, prefetchcachelines(192, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset27-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset28:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset28-postdominated_prefetch), null, prefetchcachelines(224, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset28-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset29:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset29-postdominated_prefetch), null, prefetchcachelines(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset29-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset30:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset30-postdominated_prefetch), null, prefetchcachelines(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset30-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset31:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset31-postdominated_prefetch), null, prefetchcachelines(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset31-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset32:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset32-postdominated_prefetch), null, prefetchcachelines(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset32-postdominated_prefetch)
+; GFX1250-NEXT: .Lpref_inst_offset33:
+; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset33-postdominated_prefetch), null, prefetchcachelines(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset33-postdominated_prefetch)
; GFX1250-NEXT: ;;#ASMSTART
; GFX1250-NEXT: .space 32000
; GFX1250-NEXT: ;;#ASMEND
@@ -251,6 +368,66 @@ define amdgpu_kernel void @postdominated_prefetch() {
; NO-PREFETCH-NEXT: .space 32000
; NO-PREFETCH-NEXT: ;;#ASMEND
; NO-PREFETCH-NEXT: s_endpgm
+;
+; GFX1250-DIST-LABEL: postdominated_prefetch:
+; GFX1250-DIST: ; %bb.0: ; %entry
+; GFX1250-DIST-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-DIST-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-DIST-NEXT: v_nop
+; GFX1250-DIST-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-DIST-NEXT: s_bfe_u32 s0, ttmp6, 0x4000c
+; GFX1250-DIST-NEXT: s_and_b32 s1, ttmp6, 15
+; GFX1250-DIST-NEXT: s_add_co_i32 s0, s0, 1
+; GFX1250-DIST-NEXT: s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
+; GFX1250-DIST-NEXT: s_mul_i32 s0, ttmp9, s0
+; GFX1250-DIST-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-DIST-NEXT: s_add_co_i32 s1, s1, s0
+; GFX1250-DIST-NEXT: s_cmp_eq_u32 s2, 0
+; GFX1250-DIST-NEXT: s_cselect_b32 s0, ttmp9, s1
+; GFX1250-DIST-NEXT: s_cmp_lg_u32 s0, 0
+; GFX1250-DIST-NEXT: s_mov_b32 s0, 0
+; GFX1250-DIST-NEXT: s_cbranch_scc0 .LBB3_2
+; GFX1250-DIST-NEXT: ; %bb.1: ; %else
+; GFX1250-DIST-NEXT: ;;#ASMSTART
+; GFX1250-DIST-NEXT: .space 8000
+; GFX1250-DIST-NEXT: ;;#ASMEND
+; GFX1250-DIST-NEXT: s_branch .LBB3_3
+; GFX1250-DIST-NEXT: .LBB3_2:
+; GFX1250-DIST-NEXT: s_mov_b32 s0, -1
+; GFX1250-DIST-NEXT: .LBB3_3: ; %Flow
+; GFX1250-DIST-NEXT: .Lpref_inst_offset19:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset19-postdominated_prefetch), null, prefetchcachelines(128, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset19-postdominated_prefetch)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset20:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset20-postdominated_prefetch), null, prefetchcachelines(160, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset20-postdominated_prefetch)
+; GFX1250-DIST-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-DIST-NEXT: s_and_b32 s0, s0, exec_lo
+; GFX1250-DIST-NEXT: s_cselect_b32 s0, 1, 0
+; GFX1250-DIST-NEXT: s_cmp_lg_u32 s0, 1
+; GFX1250-DIST-NEXT: s_cbranch_scc1 .LBB3_5
+; GFX1250-DIST-NEXT: ; %bb.4: ; %then
+; GFX1250-DIST-NEXT: ;;#ASMSTART
+; GFX1250-DIST-NEXT: .space 8000
+; GFX1250-DIST-NEXT: ;;#ASMEND
+; GFX1250-DIST-NEXT: .LBB3_5: ; %join
+; GFX1250-DIST-NEXT: .Lpref_inst_offset21:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset21-postdominated_prefetch), null, prefetchcachelines(192, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset21-postdominated_prefetch)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset22:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset22-postdominated_prefetch), null, prefetchcachelines(224, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset22-postdominated_prefetch)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset23:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch), null, prefetchcachelines(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset24:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch), null, prefetchcachelines(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset25:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset25-postdominated_prefetch), null, prefetchcachelines(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset25-postdominated_prefetch)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset26:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset26-postdominated_prefetch), null, prefetchcachelines(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset26-postdominated_prefetch)
+; GFX1250-DIST-NEXT: .Lpref_inst_offset27:
+; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset27-postdominated_prefetch), null, prefetchcachelines(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset27-postdominated_prefetch)
+; GFX1250-DIST-NEXT: ;;#ASMSTART
+; GFX1250-DIST-NEXT: .space 32000
+; GFX1250-DIST-NEXT: ;;#ASMEND
+; GFX1250-DIST-NEXT: s_endpgm
+; GFX1250-DIST-NEXT: .Lpref_func_end2:
entry:
%id = call i32 @llvm.amdgcn.workgroup.id.x()
%cond = icmp eq i32 %id, 0
@@ -289,6 +466,15 @@ define void @called_function() {
; NO-PREFETCH-NEXT: .space 40000
; NO-PREFETCH-NEXT: ;;#ASMEND
; NO-PREFETCH-NEXT: s_set_pc_i64 s[30:31]
+;
+; GFX1250-DIST-LABEL: called_function:
+; GFX1250-DIST: ; %bb.0:
+; GFX1250-DIST-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-DIST-NEXT: s_wait_kmcnt 0x0
+; GFX1250-DIST-NEXT: ;;#ASMSTART
+; GFX1250-DIST-NEXT: .space 40000
+; GFX1250-DIST-NEXT: ;;#ASMEND
+; GFX1250-DIST-NEXT: s_set_pc_i64 s[30:31]
call void asm sideeffect ".space 40000", ""()
ret void
}
>From 0d9a6b8d8c03837a87c1bcbfab1f5672f466c4a6 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Wed, 30 Sep 2026 09:02:42 -0500
Subject: [PATCH 17/22] Update BB prologue matcher
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
.../AMDGPU/AMDGPUInsertICachePrefetch.cpp | 12 +--
.../CodeGen/AMDGPU/icache-prefetch-config.ll | 10 ++-
llvm/test/CodeGen/AMDGPU/icache-prefetch.ll | 74 +++++++++----------
3 files changed, 52 insertions(+), 44 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index 73766a35e2ed2..be39954a6dec8 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -159,11 +159,13 @@ static MachineBasicBlock::iterator findMBBInsertionPoint(MachineBasicBlock &MBB,
continue;
}
if (!SkippedInitialUnclausedVmemPrologue &&
- InsertPt->getOpcode() == AMDGPU::GLOBAL_PREFETCH_B8_SADDR) {
- auto Next = InsertPt;
- ++Next;
- if (Next != MBB.end() && Next->getOpcode() == AMDGPU::V_NOP_e32) {
- InsertPt = ++Next;
+ InsertPt->getOpcode() == AMDGPU::S_MOV_B64) {
+ auto Vnop = std::next(InsertPt);
+ auto GlobalPrefetch = Vnop == MBB.end() ? MBB.end() : std::next(Vnop);
+ if (Vnop != MBB.end() && Vnop->getOpcode() == AMDGPU::V_NOP_e32 &&
+ GlobalPrefetch != MBB.end() &&
+ GlobalPrefetch->getOpcode() == AMDGPU::GLOBAL_PREFETCH_B8_SADDR) {
+ InsertPt = std::next(GlobalPrefetch);
SkippedInitialUnclausedVmemPrologue = true;
continue;
}
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
index 574ca53a26363..edb853f4f2d64 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
@@ -68,11 +68,17 @@ define amdgpu_kernel void @between_override_thresholds() {
}
; INITIAL-16K-LABEL: cache_size_limit:
+; INITIAL-16K: s_mov_b64
+; INITIAL-16K-NEXT: v_nop
+; INITIAL-16K-NEXT: global_prefetch_b8
; INITIAL-16K-COUNT-12: s_prefetch_inst_pc_rel
-; INITIAL-16K-NEXT: s_mov_b64
+; INITIAL-16K-NEXT: ;;#ASMSTART
; GFX1250-LABEL: cache_size_limit:
+; GFX1250: s_mov_b64
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8
; GFX1250-COUNT-14: s_prefetch_inst_pc_rel
-; GFX1250-NEXT: s_mov_b64
+; GFX1250-NEXT: ;;#ASMSTART
define amdgpu_kernel void @cache_size_limit() {
call void asm sideeffect ".space 65536", ""()
ret void
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 286cfb8cb5949..628579dcbcacb 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -58,19 +58,22 @@ define amdgpu_kernel void @below_threshold() {
; The descriptor prefetches the first 8KiB; explicit requests cover the
; remaining code in the 64KiB I-cache window.
; GFX1250-OBJ-LABEL: <partial_final_slot>:
-; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x1ff8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2ff0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3fe8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4fe0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5fd8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6fd0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7fc8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8fc0, null, 25
+; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x1fe4, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2fdc, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3fd4, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4fcc, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5fc4, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6fbc, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7fb4, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8fac, null, 25
; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x0, null, 0
define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; GFX1250-LABEL: partial_final_slot:
; GFX1250: ; %bb.0:
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: .Lpref_inst_offset0:
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(64, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(64, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
; GFX1250-NEXT: .Lpref_inst_offset1:
@@ -89,9 +92,6 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset7-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset7-partial_final_slot)
; GFX1250-NEXT: .Lpref_inst_offset8:
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset8-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset8-partial_final_slot)
-; GFX1250-NEXT: s_mov_b64 s[64:65], 0
-; GFX1250-NEXT: v_nop
-; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
; GFX1250-NEXT: v_mov_b32_e32 v0, 0
; GFX1250-NEXT: ;;#ASMSTART
@@ -120,6 +120,9 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; GFX1250-DIST-LABEL: partial_final_slot:
; GFX1250-DIST: ; %bb.0:
; GFX1250-DIST-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-DIST-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-DIST-NEXT: v_nop
+; GFX1250-DIST-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-DIST-NEXT: .Lpref_inst_offset0:
; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
; GFX1250-DIST-NEXT: .Lpref_inst_offset1:
@@ -134,9 +137,6 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot)
; GFX1250-DIST-NEXT: .Lpref_inst_offset6:
; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot)
-; GFX1250-DIST-NEXT: s_mov_b64 s[64:65], 0
-; GFX1250-DIST-NEXT: v_nop
-; GFX1250-DIST-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-DIST-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
; GFX1250-DIST-NEXT: v_mov_b32_e32 v0, 0
; GFX1250-DIST-NEXT: ;;#ASMSTART
@@ -154,24 +154,27 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; The 8KiB descriptor prefix leaves fourteen explicit requests for the complete
; 64KiB I-cache range.
; GFX1250-OBJ-LABEL: <cache_size_limit>:
-; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x1ff8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2ff0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3fe8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4fe0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5fd8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6fd0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7fc8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8fc0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9fb8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xafb0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xbfa8, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xcfa0, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xdf98, null, 31
-; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xef90, null, 31
+; GFX1250-OBJ: s_prefetch_inst_pc_rel 0x1fe4, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x2fdc, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x3fd4, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x4fcc, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x5fc4, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x6fbc, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x7fb4, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x8fac, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0x9fa4, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xaf9c, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xbf94, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xcf8c, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xdf84, null, 31
+; GFX1250-OBJ-NEXT: s_prefetch_inst_pc_rel 0xef7c, null, 31
define amdgpu_kernel void @cache_size_limit() {
; GFX1250-LABEL: cache_size_limit:
; GFX1250: ; %bb.0:
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: .Lpref_inst_offset9:
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(64, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit), null, prefetchcachelines(64, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit)
; GFX1250-NEXT: .Lpref_inst_offset10:
@@ -200,9 +203,6 @@ define amdgpu_kernel void @cache_size_limit() {
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset21-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset21-cache_size_limit)
; GFX1250-NEXT: .Lpref_inst_offset22:
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset22-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset22-cache_size_limit)
-; GFX1250-NEXT: s_mov_b64 s[64:65], 0
-; GFX1250-NEXT: v_nop
-; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: ;;#ASMSTART
; GFX1250-NEXT: .space 65536
; GFX1250-NEXT: ;;#ASMEND
@@ -223,6 +223,9 @@ define amdgpu_kernel void @cache_size_limit() {
; GFX1250-DIST-LABEL: cache_size_limit:
; GFX1250-DIST: ; %bb.0:
; GFX1250-DIST-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-DIST-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-DIST-NEXT: v_nop
+; GFX1250-DIST-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-DIST-NEXT: .Lpref_inst_offset7:
; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit), null, prefetchcachelines(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit)
; GFX1250-DIST-NEXT: .Lpref_inst_offset8:
@@ -247,9 +250,6 @@ define amdgpu_kernel void @cache_size_limit() {
; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit)
; GFX1250-DIST-NEXT: .Lpref_inst_offset18:
; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit)
-; GFX1250-DIST-NEXT: s_mov_b64 s[64:65], 0
-; GFX1250-DIST-NEXT: v_nop
-; GFX1250-DIST-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-DIST-NEXT: ;;#ASMSTART
; GFX1250-DIST-NEXT: .space 65536
; GFX1250-DIST-NEXT: ;;#ASMEND
@@ -267,13 +267,13 @@ define amdgpu_kernel void @postdominated_prefetch() {
; GFX1250-LABEL: postdominated_prefetch:
; GFX1250: ; %bb.0: ; %entry
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: .Lpref_inst_offset23:
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(64, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch), null, prefetchcachelines(64, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch)
; GFX1250-NEXT: .Lpref_inst_offset24:
; GFX1250-NEXT: s_prefetch_inst_pc_rel prefetchoffset(96, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch), null, prefetchcachelines(96, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch)
-; GFX1250-NEXT: s_mov_b64 s[64:65], 0
-; GFX1250-NEXT: v_nop
-; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: s_bfe_u32 s0, ttmp6, 0x4000c
; GFX1250-NEXT: s_and_b32 s1, ttmp6, 15
; GFX1250-NEXT: s_add_co_i32 s0, s0, 1
>From 634bab5e18ee98e4f639745947cfba070baf78ba Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 1 Oct 2026 02:06:13 -0500
Subject: [PATCH 18/22] Roundtrip ICachePrefetchLines through YAML
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp | 2 ++
llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h | 2 ++
llvm/test/CodeGen/AMDGPU/icache-prefetch.ll | 4 ++++
llvm/test/CodeGen/MIR/AMDGPU/long-branch-reg-all-sgpr-used.ll | 2 ++
.../CodeGen/MIR/AMDGPU/machine-function-info-after-pei.ll | 1 +
.../MIR/AMDGPU/machine-function-info-long-branch-reg-debug.ll | 1 +
.../MIR/AMDGPU/machine-function-info-long-branch-reg.ll | 1 +
llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-no-ir.mir | 4 ++++
llvm/test/CodeGen/MIR/AMDGPU/machine-function-info.ll | 4 ++++
9 files changed, 21 insertions(+)
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp
index d1df50d26a832..ffbf1151dae48 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp
@@ -750,6 +750,7 @@ yaml::SIMachineFunctionInfo::SIMachineFunctionInfo(
DynamicVGPRBlockSize(MFI.getDynamicVGPRBlockSize()),
ScratchReservedForDynamicVGPRs(MFI.getScratchReservedForDynamicVGPRs()),
NumKernargPreloadSGPRs(MFI.getNumKernargPreloadedSGPRs()),
+ ICachePrefetchLines(MFI.getICachePrefetchLines()),
MinNumAGPRs(MFI.getMinNumAGPRs()) {
for (Register Reg : MFI.getSGPRSpillPhysVGPRs())
SpillPhysVGPRS.push_back(regToString(Reg, TRI));
@@ -797,6 +798,7 @@ bool SIMachineFunctionInfo::initializeBaseYamlFields(
NumWaveDispatchVGPRs = YamlMFI.NumWaveDispatchVGPRs;
BytesInStackArgArea = YamlMFI.BytesInStackArgArea;
ReturnsVoid = YamlMFI.ReturnsVoid;
+ ICachePrefetchLines = YamlMFI.ICachePrefetchLines;
IsWholeWaveFunction = YamlMFI.IsWholeWaveFunction;
MinNumAGPRs = YamlMFI.MinNumAGPRs;
// This can also be set by the function attribute, MFI has higher precedence
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
index 78240ecd67e37..6e58ac3283318 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
@@ -309,6 +309,7 @@ struct SIMachineFunctionInfo final : public yaml::MachineFunctionInfo {
unsigned ScratchReservedForDynamicVGPRs = 0;
unsigned NumKernargPreloadSGPRs = 0;
+ unsigned ICachePrefetchLines = 0;
unsigned MinNumAGPRs = ~0u;
@@ -370,6 +371,7 @@ template <> struct MappingTraits<SIMachineFunctionInfo> {
YamlIO.mapOptional("scratchReservedForDynamicVGPRs",
MFI.ScratchReservedForDynamicVGPRs, 0);
YamlIO.mapOptional("numKernargPreloadSGPRs", MFI.NumKernargPreloadSGPRs, 0);
+ YamlIO.mapOptional("iCachePrefetchLines", MFI.ICachePrefetchLines, 0u);
YamlIO.mapOptional("isWholeWaveFunction", MFI.IsWholeWaveFunction, false);
YamlIO.mapOptional("minNumAGPRs", MFI.MinNumAGPRs, ~0u);
}
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 628579dcbcacb..796702bda1c53 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -11,6 +11,10 @@
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
; RUN: llvm-objdump -d %t.o | FileCheck -check-prefix=GFX1250-OBJ %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -stop-after=amdgpu-insert-icache-prefetch -o %t.mir %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -start-after=amdgpu-insert-icache-prefetch -filetype=obj -o %t-resumed.o %t.mir
+; RUN: llvm-objdump -d %t-resumed.o | FileCheck -check-prefix=GFX1250-OBJ %s
+
; Use .space to make the estimated MachineFunction size and final assembled
; size large without spelling out thousands of instructions.
diff --git a/llvm/test/CodeGen/MIR/AMDGPU/long-branch-reg-all-sgpr-used.ll b/llvm/test/CodeGen/MIR/AMDGPU/long-branch-reg-all-sgpr-used.ll
index 025008fac5144..752e353fd440c 100644
--- a/llvm/test/CodeGen/MIR/AMDGPU/long-branch-reg-all-sgpr-used.ll
+++ b/llvm/test/CodeGen/MIR/AMDGPU/long-branch-reg-all-sgpr-used.ll
@@ -49,6 +49,7 @@
; CHECK-NEXT: dynamicVGPRBlockSize: 0
; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
; CHECK-NEXT: isWholeWaveFunction: false
; CHECK-NEXT: minNumAGPRs: 4294967295
; CHECK-NEXT: body:
@@ -323,6 +324,7 @@
; CHECK-NEXT: dynamicVGPRBlockSize: 0
; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
; CHECK-NEXT: isWholeWaveFunction: false
; CHECK-NEXT: minNumAGPRs: 4294967295
; CHECK-NEXT: body:
diff --git a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-after-pei.ll b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-after-pei.ll
index 2c31f5c9c3477..6078926115f06 100644
--- a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-after-pei.ll
+++ b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-after-pei.ll
@@ -49,6 +49,7 @@
; AFTER-PEI-NEXT: dynamicVGPRBlockSize: 0
; AFTER-PEI-NEXT: scratchReservedForDynamicVGPRs: 0
; AFTER-PEI-NEXT: numKernargPreloadSGPRs: 0
+; AFTER-PEI-NEXT: iCachePrefetchLines: 0
; AFTER-PEI-NEXT: isWholeWaveFunction: false
; AFTER-PEI-NEXT: minNumAGPRs: 4294967295
; AFTER-PEI-NEXT: body:
diff --git a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg-debug.ll b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg-debug.ll
index 5497316fc1936..32e1a7ebb9a2d 100644
--- a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg-debug.ll
+++ b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg-debug.ll
@@ -49,6 +49,7 @@
; CHECK-NEXT: dynamicVGPRBlockSize: 0
; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
; CHECK-NEXT: isWholeWaveFunction: false
; CHECK-NEXT: minNumAGPRs: 4294967295
; CHECK-NEXT: body:
diff --git a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg.ll b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg.ll
index 8aab1c0fa55c8..23e3ea0425c4b 100644
--- a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg.ll
+++ b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg.ll
@@ -49,6 +49,7 @@
; CHECK-NEXT: dynamicVGPRBlockSize: 0
; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
; CHECK-NEXT: isWholeWaveFunction: false
; CHECK-NEXT: minNumAGPRs: 4294967295
; CHECK-NEXT: body:
diff --git a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-no-ir.mir b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-no-ir.mir
index 9d1236b0f3739..a6e33ca9779e5 100644
--- a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-no-ir.mir
+++ b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-no-ir.mir
@@ -58,6 +58,7 @@
# FULL-NEXT: dynamicVGPRBlockSize: 0
# FULL-NEXT: scratchReservedForDynamicVGPRs: 0
# FULL-NEXT: numKernargPreloadSGPRs: 0
+# FULL-NEXT: iCachePrefetchLines: 0
# FULL-NEXT: isWholeWaveFunction: false
# FULL-NEXT: minNumAGPRs: 4294967295
# FULL-NEXT: body:
@@ -170,6 +171,7 @@ body: |
# FULL-NEXT: dynamicVGPRBlockSize: 0
# FULL-NEXT: scratchReservedForDynamicVGPRs: 0
# FULL-NEXT: numKernargPreloadSGPRs: 0
+# FULL-NEXT: iCachePrefetchLines: 0
# FULL-NEXT: isWholeWaveFunction: false
# FULL-NEXT: minNumAGPRs: 4294967295
# FULL-NEXT: body:
@@ -254,6 +256,7 @@ body: |
# FULL-NEXT: dynamicVGPRBlockSize: 0
# FULL-NEXT: scratchReservedForDynamicVGPRs: 0
# FULL-NEXT: numKernargPreloadSGPRs: 0
+# FULL-NEXT: iCachePrefetchLines: 0
# FULL-NEXT: isWholeWaveFunction: false
# FULL-NEXT: minNumAGPRs: 4294967295
# FULL-NEXT: body:
@@ -339,6 +342,7 @@ body: |
# FULL-NEXT: dynamicVGPRBlockSize: 0
# FULL-NEXT: scratchReservedForDynamicVGPRs: 0
# FULL-NEXT: numKernargPreloadSGPRs: 0
+# FULL-NEXT: iCachePrefetchLines: 0
# FULL-NEXT: isWholeWaveFunction: false
# FULL-NEXT: minNumAGPRs: 4294967295
# FULL-NEXT: body:
diff --git a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info.ll b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info.ll
index 770718c166d4c..d38139d11a6f2 100644
--- a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info.ll
+++ b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info.ll
@@ -59,6 +59,7 @@
; CHECK-NEXT: dynamicVGPRBlockSize: 0
; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
; CHECK-NEXT: isWholeWaveFunction: false
; CHECK-NEXT: minNumAGPRs: 4294967295
; CHECK-NEXT: body:
@@ -113,6 +114,7 @@ define amdgpu_kernel void @kernel(i32 %arg0, i64 %arg1, <16 x i32> %arg2) {
; CHECK-NEXT: dynamicVGPRBlockSize: 0
; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
; CHECK-NEXT: isWholeWaveFunction: false
; CHECK-NEXT: minNumAGPRs: 4294967295
; CHECK-NEXT: body:
@@ -191,6 +193,7 @@ define amdgpu_ps void @gds_size_shader(i32 %arg0, i32 inreg %arg1) #5 {
; CHECK-NEXT: dynamicVGPRBlockSize: 0
; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
; CHECK-NEXT: isWholeWaveFunction: false
; CHECK-NEXT: minNumAGPRs: 4294967295
; CHECK-NEXT: body:
@@ -251,6 +254,7 @@ define void @function() {
; CHECK-NEXT: dynamicVGPRBlockSize: 0
; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
; CHECK-NEXT: isWholeWaveFunction: false
; CHECK-NEXT: minNumAGPRs: 4294967295
; CHECK-NEXT: body:
>From e4216b4b5b8e96c48f5004279b3b5621771aa168 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 1 Oct 2026 02:45:27 -0500
Subject: [PATCH 19/22] Fix off-by-one for INST_PREF_SIZE
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 3 +--
.../AMDGPU/AMDGPUInsertICachePrefetch.cpp | 3 ++-
.../CodeGen/AMDGPU/icache-prefetch-config.ll | 27 +++++++++++--------
3 files changed, 19 insertions(+), 14 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 3776390d8deb2..e765cd0c76993 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -273,8 +273,7 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
const MCExpr *InstPrefSize;
if (MFI.hasICachePrefetch()) {
- InstPrefSize =
- MCConstantExpr::create(MFI.getICachePrefetchLines() - 1, Ctx);
+ InstPrefSize = MCConstantExpr::create(MFI.getICachePrefetchLines(), Ctx);
} else {
const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index be39954a6dec8..196409edc872e 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -104,7 +104,8 @@ static ICachePrefetchConfig getICachePrefetchConfig(const GCNSubtarget &ST) {
uint32_t Mask, Shift, Width, CacheLineSize;
ST.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
- uint64_t DescriptorPrefetchCapacity = (uint64_t{1} << Width) * CacheLineSize;
+ uint64_t DescriptorPrefetchCapacity =
+ ((uint64_t{1} << Width) - 1) * CacheLineSize;
uint64_t InitialSize = ICachePrefetchInitialSize.getNumOccurrences()
? ICachePrefetchInitialSize
: ST.getInitialInstPrefSize();
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
index edb853f4f2d64..3006a1ca588b3 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
@@ -1,31 +1,36 @@
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -o - %s | FileCheck -check-prefix=GFX1200 %s
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -o - %s | FileCheck -check-prefix=GFX1250 %s
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -o - %s | FileCheck -check-prefix=INITIAL-16K %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=32640 -o - %s | FileCheck -check-prefix=MAX-INITIAL %s
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=49152 -o - %s | FileCheck -check-prefix=THRESHOLD-48K %s
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=THRESHOLD-16K %s
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=EQUAL-16K %s
; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=0 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=8193 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=32896 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=32768 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=0 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=32769 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=65664 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=8192 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-ORDER %s
-; The default threshold is the 32KiB descriptor capacity on both targets.
+; The default threshold is the 32640-byte descriptor capacity on both targets.
; gfx1200 initially prefetches 16KiB, while gfx1250 initially prefetches 8KiB.
define amdgpu_kernel void @above_default_threshold() {
; GFX1200-LABEL: above_default_threshold:
; GFX1200: s_prefetch_inst_pc_rel prefetchoffset(128,
-; GFX1200: .amdhsa_inst_pref_size 127
+; GFX1200: .amdhsa_inst_pref_size 128
;
; GFX1250-LABEL: above_default_threshold:
; GFX1250: s_prefetch_inst_pc_rel prefetchoffset(64,
-; GFX1250: .amdhsa_inst_pref_size 63
+; GFX1250: .amdhsa_inst_pref_size 64
;
; INITIAL-16K-LABEL: above_default_threshold:
; INITIAL-16K: s_prefetch_inst_pc_rel prefetchoffset(128,
-; INITIAL-16K: .amdhsa_inst_pref_size 127
+; INITIAL-16K: .amdhsa_inst_pref_size 128
+;
+; MAX-INITIAL-LABEL: above_default_threshold:
+; MAX-INITIAL: s_prefetch_inst_pc_rel prefetchoffset(255,
+; MAX-INITIAL: .amdhsa_inst_pref_size 255
;
; THRESHOLD-48K-LABEL: above_default_threshold:
; THRESHOLD-48K-NOT: s_prefetch_inst_pc_rel
@@ -36,19 +41,19 @@ define amdgpu_kernel void @above_default_threshold() {
}
; The gfx1250 prologue and s_endpgm add 32 bytes, making these functions
-; exactly 32KiB and four bytes over 32KiB respectively.
+; exactly 32640 bytes and four bytes over 32640 bytes respectively.
define amdgpu_kernel void @at_default_threshold() {
; GFX1250-LABEL: at_default_threshold:
; GFX1250-NOT: s_prefetch_inst_pc_rel
; GFX1250: s_endpgm
- call void asm sideeffect ".space 32736", ""()
+ call void asm sideeffect ".space 32608", ""()
ret void
}
define amdgpu_kernel void @above_default_threshold_by_four() {
; GFX1250-LABEL: above_default_threshold_by_four:
; GFX1250: s_prefetch_inst_pc_rel prefetchoffset(64,
- call void asm sideeffect ".space 32740", ""()
+ call void asm sideeffect ".space 32612", ""()
ret void
}
@@ -58,11 +63,11 @@ define amdgpu_kernel void @above_default_threshold_by_four() {
define amdgpu_kernel void @between_override_thresholds() {
; THRESHOLD-16K-LABEL: between_override_thresholds:
; THRESHOLD-16K: s_prefetch_inst_pc_rel prefetchoffset(64,
-; THRESHOLD-16K: .amdhsa_inst_pref_size 63
+; THRESHOLD-16K: .amdhsa_inst_pref_size 64
;
; EQUAL-16K-LABEL: between_override_thresholds:
; EQUAL-16K: s_prefetch_inst_pc_rel prefetchoffset(128,
-; EQUAL-16K: .amdhsa_inst_pref_size 127
+; EQUAL-16K: .amdhsa_inst_pref_size 128
call void asm sideeffect ".space 20000", ""()
ret void
}
@@ -84,6 +89,6 @@ define amdgpu_kernel void @cache_size_limit() {
ret void
}
-; INVALID-INITIAL: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must be a non-zero multiple of 128 bytes not exceeding 32768 bytes for gfx1250
+; INVALID-INITIAL: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must be a non-zero multiple of 128 bytes not exceeding 32640 bytes for gfx1250
; INVALID-THRESHOLD: LLVM ERROR: -amdgpu-icache-prefetch-threshold must be a non-zero multiple of 128 bytes not exceeding 65536 bytes for gfx1250
; INVALID-ORDER: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must not exceed -amdgpu-icache-prefetch-threshold for gfx1250
>From 8e6947c0d48c747a33dddf8def3771e6816a242a Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 1 Oct 2026 03:25:08 -0500
Subject: [PATCH 20/22] Remove outdated code
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h | 3 ---
llvm/lib/Target/AMDGPU/SIDefines.h | 3 ---
2 files changed, 6 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
index 871bd738b947b..ca82af08ffade 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
@@ -165,9 +165,6 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
/// Get the symbol for the function end, used for ICache prefetch MCExprs.
MCSymbol *getPrefetchEndSym() const { return PrefetchEndSym; }
- /// Get the code size estimate from SIProgramInfo.
- uint64_t getCodeSize() { return CurrentProgramInfo.getFunctionCodeSize(*MF); }
-
protected:
void getAnalysisUsage(AnalysisUsage &AU) const override;
diff --git a/llvm/lib/Target/AMDGPU/SIDefines.h b/llvm/lib/Target/AMDGPU/SIDefines.h
index a67c259541378..864a84cc8daf4 100644
--- a/llvm/lib/Target/AMDGPU/SIDefines.h
+++ b/llvm/lib/Target/AMDGPU/SIDefines.h
@@ -824,9 +824,6 @@ enum ModeRegisterMasks : uint32_t {
SRC2_VGPR_MSB = 0x3 << 18,
VGPR_MSB_MASK = 0xff << 12, // Bits 12..19
- // GFX1250 instruction prefetch
- SCALAR_PREFETCH_EN = 1 << 24,
-
REPLAY_MODE = 1 << 25,
FLAT_SCRATCH_IS_NV = 1 << 26,
};
>From 83f108b2960f46c15b8dab0a9cea279be3dd8f5c Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 1 Oct 2026 03:25:44 -0500
Subject: [PATCH 21/22] Simplify tests
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
.../CodeGen/AMDGPU/icache-prefetch-config.ll | 78 +-----
llvm/test/CodeGen/AMDGPU/icache-prefetch.ll | 252 +-----------------
2 files changed, 15 insertions(+), 315 deletions(-)
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
index 3006a1ca588b3..0b5269712fb3a 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
@@ -1,62 +1,19 @@
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -o - %s | FileCheck -check-prefix=GFX1200 %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -o - %s | FileCheck -check-prefix=GFX1250 %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -o - %s | FileCheck -check-prefix=INITIAL-16K %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=32640 -o - %s | FileCheck -check-prefix=MAX-INITIAL %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=49152 -o - %s | FileCheck -check-prefix=THRESHOLD-48K %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=THRESHOLD-16K %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=EQUAL-16K %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=0 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=8193 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=32768 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=0 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=32769 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=65664 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=8192 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-ORDER %s
+; Check the gfx1200 default initial prefetch size.
+; RUN: llc -mtriple=amdgpu12.00-amd-amdhsa -o - %s | FileCheck -check-prefix=GFX1200 %s
+; Check overriding only the explicit-prefetch threshold.
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=THRESHOLD-16K %s
+; Check independently overriding the initial prefetch size.
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=EQUAL-16K %s
-; The default threshold is the 32640-byte descriptor capacity on both targets.
-; gfx1200 initially prefetches 16KiB, while gfx1250 initially prefetches 8KiB.
+; The default threshold is the 32640-byte descriptor capacity.
define amdgpu_kernel void @above_default_threshold() {
; GFX1200-LABEL: above_default_threshold:
; GFX1200: s_prefetch_inst_pc_rel prefetchoffset(128,
; GFX1200: .amdhsa_inst_pref_size 128
-;
-; GFX1250-LABEL: above_default_threshold:
-; GFX1250: s_prefetch_inst_pc_rel prefetchoffset(64,
-; GFX1250: .amdhsa_inst_pref_size 64
-;
-; INITIAL-16K-LABEL: above_default_threshold:
-; INITIAL-16K: s_prefetch_inst_pc_rel prefetchoffset(128,
-; INITIAL-16K: .amdhsa_inst_pref_size 128
-;
-; MAX-INITIAL-LABEL: above_default_threshold:
-; MAX-INITIAL: s_prefetch_inst_pc_rel prefetchoffset(255,
-; MAX-INITIAL: .amdhsa_inst_pref_size 255
-;
-; THRESHOLD-48K-LABEL: above_default_threshold:
-; THRESHOLD-48K-NOT: s_prefetch_inst_pc_rel
-; THRESHOLD-48K: s_endpgm
-; THRESHOLD-48K: .amdhsa_inst_pref_size ((instprefsize(
call void asm sideeffect ".space 40000", ""()
ret void
}
-; The gfx1250 prologue and s_endpgm add 32 bytes, making these functions
-; exactly 32640 bytes and four bytes over 32640 bytes respectively.
-define amdgpu_kernel void @at_default_threshold() {
-; GFX1250-LABEL: at_default_threshold:
-; GFX1250-NOT: s_prefetch_inst_pc_rel
-; GFX1250: s_endpgm
- call void asm sideeffect ".space 32608", ""()
- ret void
-}
-
-define amdgpu_kernel void @above_default_threshold_by_four() {
-; GFX1250-LABEL: above_default_threshold_by_four:
-; GFX1250: s_prefetch_inst_pc_rel prefetchoffset(64,
- call void asm sideeffect ".space 32612", ""()
- ret void
-}
-
; Lowering only the threshold activates explicit prefetching while retaining
; the default 8KiB initial descriptor coverage. Setting initial == threshold
; is also valid and changes the descriptor coverage independently.
@@ -71,24 +28,3 @@ define amdgpu_kernel void @between_override_thresholds() {
call void asm sideeffect ".space 20000", ""()
ret void
}
-
-; INITIAL-16K-LABEL: cache_size_limit:
-; INITIAL-16K: s_mov_b64
-; INITIAL-16K-NEXT: v_nop
-; INITIAL-16K-NEXT: global_prefetch_b8
-; INITIAL-16K-COUNT-12: s_prefetch_inst_pc_rel
-; INITIAL-16K-NEXT: ;;#ASMSTART
-; GFX1250-LABEL: cache_size_limit:
-; GFX1250: s_mov_b64
-; GFX1250-NEXT: v_nop
-; GFX1250-NEXT: global_prefetch_b8
-; GFX1250-COUNT-14: s_prefetch_inst_pc_rel
-; GFX1250-NEXT: ;;#ASMSTART
-define amdgpu_kernel void @cache_size_limit() {
- call void asm sideeffect ".space 65536", ""()
- ret void
-}
-
-; INVALID-INITIAL: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must be a non-zero multiple of 128 bytes not exceeding 32640 bytes for gfx1250
-; INVALID-THRESHOLD: LLVM ERROR: -amdgpu-icache-prefetch-threshold must be a non-zero multiple of 128 bytes not exceeding 65536 bytes for gfx1250
-; INVALID-ORDER: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must not exceed -amdgpu-icache-prefetch-threshold for gfx1250
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 796702bda1c53..39c22d3d59722 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -1,18 +1,15 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -o - %s | \
+; Check prefetch insertion and distribution.
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -amdgpu-icache-prefetch=true -o - %s | \
; RUN: FileCheck -check-prefix=GFX1250 %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=false -o - %s | \
-; RUN: FileCheck -check-prefix=NO-PREFETCH %s
-
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -amdgpu-icache-prefetch-initial-size=16384 -o - %s | \
-; RUN: FileCheck -check-prefix=GFX1250-DIST %s
-
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
+; Check final prefetch offsets and cache-line counts.
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
; RUN: llvm-objdump -d %t.o | FileCheck -check-prefix=GFX1250-OBJ %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -stop-after=amdgpu-insert-icache-prefetch -o %t.mir %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -start-after=amdgpu-insert-icache-prefetch -filetype=obj -o %t-resumed.o %t.mir
+; Check prefetch state survives a MIR round trip.
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -amdgpu-icache-prefetch=true -stop-after=amdgpu-insert-icache-prefetch -o %t.mir %s
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -start-after=amdgpu-insert-icache-prefetch -filetype=obj -o %t-resumed.o %t.mir
; RUN: llvm-objdump -d %t-resumed.o | FileCheck -check-prefix=GFX1250-OBJ %s
; Use .space to make the estimated MachineFunction size and final assembled
@@ -33,28 +30,6 @@ define amdgpu_kernel void @below_threshold() {
; GFX1250-NEXT: .space 30000
; GFX1250-NEXT: ;;#ASMEND
; GFX1250-NEXT: s_endpgm
-;
-; NO-PREFETCH-LABEL: below_threshold:
-; NO-PREFETCH: ; %bb.0:
-; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; NO-PREFETCH-NEXT: s_mov_b64 s[64:65], 0
-; NO-PREFETCH-NEXT: v_nop
-; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; NO-PREFETCH-NEXT: ;;#ASMSTART
-; NO-PREFETCH-NEXT: .space 30000
-; NO-PREFETCH-NEXT: ;;#ASMEND
-; NO-PREFETCH-NEXT: s_endpgm
-;
-; GFX1250-DIST-LABEL: below_threshold:
-; GFX1250-DIST: ; %bb.0:
-; GFX1250-DIST-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-DIST-NEXT: s_mov_b64 s[64:65], 0
-; GFX1250-DIST-NEXT: v_nop
-; GFX1250-DIST-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX1250-DIST-NEXT: ;;#ASMSTART
-; GFX1250-DIST-NEXT: .space 30000
-; GFX1250-DIST-NEXT: ;;#ASMEND
-; GFX1250-DIST-NEXT: s_endpgm
call void asm sideeffect ".space 30000", ""()
ret void
}
@@ -105,51 +80,6 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
; GFX1250-NEXT: global_store_b32 v0, v0, s[0:1]
; GFX1250-NEXT: s_endpgm
; GFX1250-NEXT: .Lpref_func_end0:
-;
-; NO-PREFETCH-LABEL: partial_final_slot:
-; NO-PREFETCH: ; %bb.0:
-; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; NO-PREFETCH-NEXT: s_mov_b64 s[64:65], 0
-; NO-PREFETCH-NEXT: v_nop
-; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; NO-PREFETCH-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
-; NO-PREFETCH-NEXT: v_mov_b32_e32 v0, 0
-; NO-PREFETCH-NEXT: ;;#ASMSTART
-; NO-PREFETCH-NEXT: .space 40000
-; NO-PREFETCH-NEXT: ;;#ASMEND
-; NO-PREFETCH-NEXT: s_wait_kmcnt 0x0
-; NO-PREFETCH-NEXT: global_store_b32 v0, v0, s[0:1]
-; NO-PREFETCH-NEXT: s_endpgm
-;
-; GFX1250-DIST-LABEL: partial_final_slot:
-; GFX1250-DIST: ; %bb.0:
-; GFX1250-DIST-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-DIST-NEXT: s_mov_b64 s[64:65], 0
-; GFX1250-DIST-NEXT: v_nop
-; GFX1250-DIST-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX1250-DIST-NEXT: .Lpref_inst_offset0:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset1:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot), null, prefetchcachelines(160, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset2:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot), null, prefetchcachelines(192, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset3:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset3-partial_final_slot), null, prefetchcachelines(224, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset3-partial_final_slot)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset4:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset4-partial_final_slot), null, prefetchcachelines(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset4-partial_final_slot)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset5:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset6:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot)
-; GFX1250-DIST-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
-; GFX1250-DIST-NEXT: v_mov_b32_e32 v0, 0
-; GFX1250-DIST-NEXT: ;;#ASMSTART
-; GFX1250-DIST-NEXT: .space 40000
-; GFX1250-DIST-NEXT: ;;#ASMEND
-; GFX1250-DIST-NEXT: s_wait_kmcnt 0x0
-; GFX1250-DIST-NEXT: global_store_b32 v0, v0, s[0:1]
-; GFX1250-DIST-NEXT: s_endpgm
-; GFX1250-DIST-NEXT: .Lpref_func_end0:
call void asm sideeffect ".space 40000", ""()
store i32 0, ptr addrspace(1) %out
ret void
@@ -212,58 +142,11 @@ define amdgpu_kernel void @cache_size_limit() {
; GFX1250-NEXT: ;;#ASMEND
; GFX1250-NEXT: s_endpgm
; GFX1250-NEXT: .Lpref_func_end1:
-;
-; NO-PREFETCH-LABEL: cache_size_limit:
-; NO-PREFETCH: ; %bb.0:
-; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; NO-PREFETCH-NEXT: s_mov_b64 s[64:65], 0
-; NO-PREFETCH-NEXT: v_nop
-; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; NO-PREFETCH-NEXT: ;;#ASMSTART
-; NO-PREFETCH-NEXT: .space 65536
-; NO-PREFETCH-NEXT: ;;#ASMEND
-; NO-PREFETCH-NEXT: s_endpgm
-;
-; GFX1250-DIST-LABEL: cache_size_limit:
-; GFX1250-DIST: ; %bb.0:
-; GFX1250-DIST-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-DIST-NEXT: s_mov_b64 s[64:65], 0
-; GFX1250-DIST-NEXT: v_nop
-; GFX1250-DIST-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX1250-DIST-NEXT: .Lpref_inst_offset7:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit), null, prefetchcachelines(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset8:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit), null, prefetchcachelines(160, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset9:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit), null, prefetchcachelines(192, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset10:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit), null, prefetchcachelines(224, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset11:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset11-cache_size_limit), null, prefetchcachelines(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset11-cache_size_limit)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset12:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset12-cache_size_limit), null, prefetchcachelines(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset12-cache_size_limit)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset13:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset13-cache_size_limit), null, prefetchcachelines(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset13-cache_size_limit)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset14:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset14-cache_size_limit), null, prefetchcachelines(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset14-cache_size_limit)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset15:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset15-cache_size_limit), null, prefetchcachelines(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset15-cache_size_limit)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset16:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset16-cache_size_limit), null, prefetchcachelines(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset16-cache_size_limit)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset17:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset18:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit)
-; GFX1250-DIST-NEXT: ;;#ASMSTART
-; GFX1250-DIST-NEXT: .space 65536
-; GFX1250-DIST-NEXT: ;;#ASMEND
-; GFX1250-DIST-NEXT: s_endpgm
-; GFX1250-DIST-NEXT: .Lpref_func_end1:
call void asm sideeffect ".space 65536", ""()
ret void
}
-; With a 16KiB descriptor prefix override, explicit prefetches are deferred to
+; Explicit prefetches are distributed along the post-dominator chain through
; the shared exit block.
declare i32 @llvm.amdgcn.workgroup.id.x()
@@ -331,107 +214,6 @@ define amdgpu_kernel void @postdominated_prefetch() {
; GFX1250-NEXT: ;;#ASMEND
; GFX1250-NEXT: s_endpgm
; GFX1250-NEXT: .Lpref_func_end2:
-;
-; NO-PREFETCH-LABEL: postdominated_prefetch:
-; NO-PREFETCH: ; %bb.0: ; %entry
-; NO-PREFETCH-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; NO-PREFETCH-NEXT: s_mov_b64 s[64:65], 0
-; NO-PREFETCH-NEXT: v_nop
-; NO-PREFETCH-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; NO-PREFETCH-NEXT: s_bfe_u32 s0, ttmp6, 0x4000c
-; NO-PREFETCH-NEXT: s_and_b32 s1, ttmp6, 15
-; NO-PREFETCH-NEXT: s_add_co_i32 s0, s0, 1
-; NO-PREFETCH-NEXT: s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
-; NO-PREFETCH-NEXT: s_mul_i32 s0, ttmp9, s0
-; NO-PREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
-; NO-PREFETCH-NEXT: s_add_co_i32 s1, s1, s0
-; NO-PREFETCH-NEXT: s_cmp_eq_u32 s2, 0
-; NO-PREFETCH-NEXT: s_cselect_b32 s0, ttmp9, s1
-; NO-PREFETCH-NEXT: s_cmp_lg_u32 s0, 0
-; NO-PREFETCH-NEXT: s_mov_b32 s0, 0
-; NO-PREFETCH-NEXT: s_cbranch_scc0 .LBB3_2
-; NO-PREFETCH-NEXT: ; %bb.1: ; %else
-; NO-PREFETCH-NEXT: ;;#ASMSTART
-; NO-PREFETCH-NEXT: .space 8000
-; NO-PREFETCH-NEXT: ;;#ASMEND
-; NO-PREFETCH-NEXT: s_branch .LBB3_3
-; NO-PREFETCH-NEXT: .LBB3_2:
-; NO-PREFETCH-NEXT: s_mov_b32 s0, -1
-; NO-PREFETCH-NEXT: .LBB3_3: ; %Flow
-; NO-PREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
-; NO-PREFETCH-NEXT: s_and_b32 s0, s0, exec_lo
-; NO-PREFETCH-NEXT: s_cselect_b32 s0, 1, 0
-; NO-PREFETCH-NEXT: s_cmp_lg_u32 s0, 1
-; NO-PREFETCH-NEXT: s_cbranch_scc1 .LBB3_5
-; NO-PREFETCH-NEXT: ; %bb.4: ; %then
-; NO-PREFETCH-NEXT: ;;#ASMSTART
-; NO-PREFETCH-NEXT: .space 8000
-; NO-PREFETCH-NEXT: ;;#ASMEND
-; NO-PREFETCH-NEXT: .LBB3_5: ; %join
-; NO-PREFETCH-NEXT: ;;#ASMSTART
-; NO-PREFETCH-NEXT: .space 32000
-; NO-PREFETCH-NEXT: ;;#ASMEND
-; NO-PREFETCH-NEXT: s_endpgm
-;
-; GFX1250-DIST-LABEL: postdominated_prefetch:
-; GFX1250-DIST: ; %bb.0: ; %entry
-; GFX1250-DIST-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-DIST-NEXT: s_mov_b64 s[64:65], 0
-; GFX1250-DIST-NEXT: v_nop
-; GFX1250-DIST-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX1250-DIST-NEXT: s_bfe_u32 s0, ttmp6, 0x4000c
-; GFX1250-DIST-NEXT: s_and_b32 s1, ttmp6, 15
-; GFX1250-DIST-NEXT: s_add_co_i32 s0, s0, 1
-; GFX1250-DIST-NEXT: s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
-; GFX1250-DIST-NEXT: s_mul_i32 s0, ttmp9, s0
-; GFX1250-DIST-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
-; GFX1250-DIST-NEXT: s_add_co_i32 s1, s1, s0
-; GFX1250-DIST-NEXT: s_cmp_eq_u32 s2, 0
-; GFX1250-DIST-NEXT: s_cselect_b32 s0, ttmp9, s1
-; GFX1250-DIST-NEXT: s_cmp_lg_u32 s0, 0
-; GFX1250-DIST-NEXT: s_mov_b32 s0, 0
-; GFX1250-DIST-NEXT: s_cbranch_scc0 .LBB3_2
-; GFX1250-DIST-NEXT: ; %bb.1: ; %else
-; GFX1250-DIST-NEXT: ;;#ASMSTART
-; GFX1250-DIST-NEXT: .space 8000
-; GFX1250-DIST-NEXT: ;;#ASMEND
-; GFX1250-DIST-NEXT: s_branch .LBB3_3
-; GFX1250-DIST-NEXT: .LBB3_2:
-; GFX1250-DIST-NEXT: s_mov_b32 s0, -1
-; GFX1250-DIST-NEXT: .LBB3_3: ; %Flow
-; GFX1250-DIST-NEXT: .Lpref_inst_offset19:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset19-postdominated_prefetch), null, prefetchcachelines(128, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset19-postdominated_prefetch)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset20:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset20-postdominated_prefetch), null, prefetchcachelines(160, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset20-postdominated_prefetch)
-; GFX1250-DIST-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
-; GFX1250-DIST-NEXT: s_and_b32 s0, s0, exec_lo
-; GFX1250-DIST-NEXT: s_cselect_b32 s0, 1, 0
-; GFX1250-DIST-NEXT: s_cmp_lg_u32 s0, 1
-; GFX1250-DIST-NEXT: s_cbranch_scc1 .LBB3_5
-; GFX1250-DIST-NEXT: ; %bb.4: ; %then
-; GFX1250-DIST-NEXT: ;;#ASMSTART
-; GFX1250-DIST-NEXT: .space 8000
-; GFX1250-DIST-NEXT: ;;#ASMEND
-; GFX1250-DIST-NEXT: .LBB3_5: ; %join
-; GFX1250-DIST-NEXT: .Lpref_inst_offset21:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset21-postdominated_prefetch), null, prefetchcachelines(192, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset21-postdominated_prefetch)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset22:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset22-postdominated_prefetch), null, prefetchcachelines(224, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset22-postdominated_prefetch)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset23:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch), null, prefetchcachelines(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset24:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch), null, prefetchcachelines(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset25:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset25-postdominated_prefetch), null, prefetchcachelines(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset25-postdominated_prefetch)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset26:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset26-postdominated_prefetch), null, prefetchcachelines(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset26-postdominated_prefetch)
-; GFX1250-DIST-NEXT: .Lpref_inst_offset27:
-; GFX1250-DIST-NEXT: s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset27-postdominated_prefetch), null, prefetchcachelines(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset27-postdominated_prefetch)
-; GFX1250-DIST-NEXT: ;;#ASMSTART
-; GFX1250-DIST-NEXT: .space 32000
-; GFX1250-DIST-NEXT: ;;#ASMEND
-; GFX1250-DIST-NEXT: s_endpgm
-; GFX1250-DIST-NEXT: .Lpref_func_end2:
entry:
%id = call i32 @llvm.amdgcn.workgroup.id.x()
%cond = icmp eq i32 %id, 0
@@ -461,24 +243,6 @@ define void @called_function() {
; GFX1250-NEXT: .space 40000
; GFX1250-NEXT: ;;#ASMEND
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
-;
-; NO-PREFETCH-LABEL: called_function:
-; NO-PREFETCH: ; %bb.0:
-; NO-PREFETCH-NEXT: s_wait_loadcnt_dscnt 0x0
-; NO-PREFETCH-NEXT: s_wait_kmcnt 0x0
-; NO-PREFETCH-NEXT: ;;#ASMSTART
-; NO-PREFETCH-NEXT: .space 40000
-; NO-PREFETCH-NEXT: ;;#ASMEND
-; NO-PREFETCH-NEXT: s_set_pc_i64 s[30:31]
-;
-; GFX1250-DIST-LABEL: called_function:
-; GFX1250-DIST: ; %bb.0:
-; GFX1250-DIST-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-DIST-NEXT: s_wait_kmcnt 0x0
-; GFX1250-DIST-NEXT: ;;#ASMSTART
-; GFX1250-DIST-NEXT: .space 40000
-; GFX1250-DIST-NEXT: ;;#ASMEND
-; GFX1250-DIST-NEXT: s_set_pc_i64 s[30:31]
call void asm sideeffect ".space 40000", ""()
ret void
}
>From f6056bd861735e8a65f9a2f1692d997f9ed0f0e2 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 1 Oct 2026 03:48:28 -0500
Subject: [PATCH 22/22] Code formatting
Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h | 4 +---
1 file changed, 1 insertion(+), 3 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
index 6e58ac3283318..e17f350ea2ef9 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
@@ -1137,9 +1137,7 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
unsigned getICachePrefetchLines() const { return ICachePrefetchLines; }
- void setICachePrefetchLines(unsigned Lines) {
- ICachePrefetchLines = Lines;
- }
+ void setICachePrefetchLines(unsigned Lines) { ICachePrefetchLines = Lines; }
unsigned getNumSpilledSGPRs() const {
return NumSpilledSGPRs;
More information about the llvm-commits
mailing list