[llvm] [AMDGPU] Explicit instruction prefetch for large entry functions (PR #227995)

Lukas Sommer via llvm-commits llvm-commits at lists.llvm.org
Thu Oct 1 01:48:57 PDT 2026


https://github.com/sommerlukas updated https://github.com/llvm/llvm-project/pull/227995

>From c552505d33361969a091f24f47797a2cb39e6e5d Mon Sep 17 00:00:00 2001
From: Jeffrey Byrnes <Jeffrey.Byrnes at amd.com>
Date: Wed, 8 Jul 2026 15:35:04 -0700
Subject: [PATCH 01/22] [AMDGPU] Prefetch up to 64kb ICache with
 s_prefetch_inst_pcrel

---
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp   |  34 +++++-
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h     |  12 ++
 llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp  |  37 ++++++
 .../AMDGPU/AsmParser/AMDGPUAsmParser.cpp      |  23 ++++
 .../AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp  |  20 ++++
 .../AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h    |   8 ++
 .../AMDGPU/MCTargetDesc/AMDGPUInstPrinter.cpp |  18 ++-
 .../AMDGPU/MCTargetDesc/AMDGPUInstPrinter.h   |   2 +
 .../MCTargetDesc/AMDGPUMCCodeEmitter.cpp      |  37 +++++-
 .../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp      |  90 ++++++++++++++
 .../Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h |  21 ++++
 llvm/lib/Target/AMDGPU/SIDefines.h            |   3 +
 .../lib/Target/AMDGPU/SIMachineFunctionInfo.h |  10 ++
 llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp  | 101 +++++++++++++++-
 llvm/lib/Target/AMDGPU/SMInstructions.td      |   5 +-
 llvm/test/CodeGen/AMDGPU/icache-prefetch.ll   | 112 ++++++++++++++++++
 16 files changed, 518 insertions(+), 15 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/icache-prefetch.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index b24f62fa69a11..7ca9ebd5db622 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -193,6 +193,11 @@ void AMDGPUAsmPrinter::emitFunctionBodyStart() {
   const GCNSubtarget &STM = MF->getSubtarget<GCNSubtarget>();
   const Function &F = MF->getFunction();
 
+  // If ICache prefetch is enabled, create the function end symbol early so
+  // it can be referenced by the prefetch MCExprs during instruction emission.
+  if (MFI.hasICachePrefetch())
+    PrefetchEndSym = createTempSymbol("pref_func_end");
+
   // TODO: We're checking this late, would be nice to check it earlier.
   if (STM.requiresCodeObjectV6() && CodeObjectVersion < AMDGPU::AMDHSA_COV6) {
     reportFatalUsageError(
@@ -220,6 +225,15 @@ void AMDGPUAsmPrinter::emitFunctionBodyStart() {
     HSAMetadataStream->emitKernel(*MF, CurrentProgramInfo);
 }
 
+void AMDGPUAsmPrinter::emitFunctionBodyEnd() {
+  // Emit the label for the prefetch end symbol if ICache prefetch is enabled.
+  // This symbol was created in emitFunctionBodyStart for this function.
+  if (PrefetchEndSym) {
+    OutStreamer->emitLabel(PrefetchEndSym);
+    PrefetchEndSym = nullptr;
+  }
+}
+
 /// Set bits in a kernel descriptor MCExpr field:
 ///   return ((Dst & ~Mask) | (Value << Shift))
 static const MCExpr *setBits(const MCExpr *Dst, const MCExpr *Value,
@@ -249,15 +263,23 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
   // size. At this point .Lfunc_end has been emitted (by the base AsmPrinter)
   // right after the function code, so (Lfunc_end - func_sym) gives the
   // exact function code size in bytes.
+  //
+  // When ICache prefetch is enabled, we set INST_PREF_SIZE to 0 because
+  // the s_prefetch_inst instructions handle all prefetching.
   if (STM.hasInstPrefSize()) {
-    const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
-        MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
-        MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
-
     uint32_t Mask, Shift, Width, CacheLineSize;
     STM.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
-    const MCExpr *InstPrefSize =
-        AMDGPUMCExpr::createInstPrefSize(CodeSizeExpr, Ctx);
+
+    const MCExpr *InstPrefSize;
+    if (MFI.hasICachePrefetch()) {
+      // Disable hardware prefetch - s_prefetch_inst handles it.
+      InstPrefSize = MCConstantExpr::create(0, Ctx);
+    } else {
+      const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
+          MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
+          MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
+      InstPrefSize = AMDGPUMCExpr::createInstPrefSize(CodeSizeExpr, Ctx);
+    }
     KD.compute_pgm_rsrc3 =
         setBits(KD.compute_pgm_rsrc3, InstPrefSize, Mask, Shift, Ctx);
   }
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
index 4394cde308665..871bd738b947b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
@@ -57,6 +57,10 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
 
   MCCodeEmitter *DumpCodeInstEmitter = nullptr;
 
+  // Symbol for the function end, used by ICache prefetch MCExprs.
+  // Created early in emitFunctionBodyStart when prefetch is enabled.
+  MCSymbol *PrefetchEndSym = nullptr;
+
   // When appropriate, add a _dvgpr$ symbol.
   void emitDVgprSymbol(MachineFunction &MF);
 
@@ -139,6 +143,8 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
 
   void emitFunctionBodyStart() override;
 
+  void emitFunctionBodyEnd() override;
+
   void endFunction(const MachineFunction *MF);
 
   void emitImplicitDef(const MachineInstr *MI) const override;
@@ -156,6 +162,12 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
   bool PrintAsmOperand(const MachineInstr *MI, unsigned OpNo,
                        const char *ExtraCode, raw_ostream &O) override;
 
+  /// Get the symbol for the function end, used for ICache prefetch MCExprs.
+  MCSymbol *getPrefetchEndSym() const { return PrefetchEndSym; }
+
+  /// Get the code size estimate from SIProgramInfo.
+  uint64_t getCodeSize() { return CurrentProgramInfo.getFunctionCodeSize(*MF); }
+
 protected:
   void getAnalysisUsage(AnalysisUsage &AU) const override;
 
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
index 1758db77d8a6e..fd4651a41f8a8 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
@@ -466,6 +466,43 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
 
     MCInst TmpInst;
     MCInstLowering.lower(MI, TmpInst);
+
+    // Fix up S_PREFETCH_INST_PC_REL instructions inserted by the ICache
+    // prefetch pass. Replace the slot index in the sdata operand with an
+    // MCExpr that computes the cacheline count based on exact code size.
+    if (MI->getOpcode() == AMDGPU::S_PREFETCH_INST_PC_REL) {
+      const SIMachineFunctionInfo *MFI = MF->getInfo<SIMachineFunctionInfo>();
+      if (MFI->hasICachePrefetch()) {
+        // Operand indices in the MCInst.
+        constexpr unsigned OffsetIdx = 0;
+        constexpr unsigned SdataIdx = 2;
+
+        // The sdata operand contains the slot index (0-15) set by the pass.
+        int64_t SlotIndex = TmpInst.getOperand(SdataIdx).getImm();
+
+        // Create MCExpr for code size using label subtraction.
+        // This gives the exact code size at assembly time.
+        const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
+            MCSymbolRefExpr::create(getPrefetchEndSym(), OutContext),
+            MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
+
+        // Create MCExpr for the slot index.
+        const MCExpr *SlotIndexExpr =
+            MCConstantExpr::create(SlotIndex, OutContext);
+
+        // Create MCExprs that will be evaluated at fixup time when symbol
+        // positions are known.
+        const MCExpr *CachelinesExpr = AMDGPUMCExpr::createPrefetchCachelines(
+            SlotIndexExpr, CodeSizeExpr, OutContext);
+        const MCExpr *OffsetExpr = AMDGPUMCExpr::createPrefetchOffset(
+            SlotIndexExpr, CodeSizeExpr, OutContext);
+
+        // Replace the offset and sdata operands with MCExprs.
+        TmpInst.getOperand(OffsetIdx) = MCOperand::createExpr(OffsetExpr);
+        TmpInst.getOperand(SdataIdx) = MCOperand::createExpr(CachelinesExpr);
+      }
+    }
+
     EmitToStreamer(*OutStreamer, TmpInst);
 
     if (DumpCodeInstEmitter) {
diff --git a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
index a01042e53bf37..9269eb15b0f5d 100644
--- a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
+++ b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
@@ -966,6 +966,7 @@ class AMDGPUOperand : public MCParsedAsmOperand {
   bool isSMRDOffset8() const;
   bool isSMEMOffset() const;
   bool isSMRDLiteralOffset() const;
+  bool isPrefetchSdata() const;
   bool isDPP8() const;
   bool isDPPCtrl() const;
   bool isBLGP() const;
@@ -9522,6 +9523,14 @@ bool AMDGPUOperand::isSMRDOffset8() const {
 
 bool AMDGPUOperand::isSMEMOffset() const {
   // Offset range is checked later by validator.
+  // Also accept prefetch offset expressions for ICache prefetch instructions.
+  if (isExpr()) {
+    if (Expr->getKind() == MCExpr::Target) {
+      const AMDGPUMCExpr *AExpr = static_cast<const AMDGPUMCExpr *>(Expr);
+      if (AExpr->getKind() == AMDGPUMCExpr::AGVK_PrefetchOffset)
+        return true;
+    }
+  }
   return isImmLiteral();
 }
 
@@ -9531,6 +9540,18 @@ bool AMDGPUOperand::isSMRDLiteralOffset() const {
   return isImmLiteral() && !isUInt<8>(getImm()) && isUInt<32>(getImm());
 }
 
+bool AMDGPUOperand::isPrefetchSdata() const {
+  // Accept immediates (u8) or prefetch cachelines expressions.
+  if (isExpr()) {
+    if (Expr->getKind() == MCExpr::Target) {
+      const AMDGPUMCExpr *AExpr = static_cast<const AMDGPUMCExpr *>(Expr);
+      if (AExpr->getKind() == AMDGPUMCExpr::AGVK_PrefetchCachelines)
+        return true;
+    }
+  }
+  return isImmLiteral() && isUInt<8>(getImm());
+}
+
 //===----------------------------------------------------------------------===//
 // vop3
 //===----------------------------------------------------------------------===//
@@ -9611,6 +9632,8 @@ bool AMDGPUAsmParser::parsePrimaryExpr(const MCExpr *&Res, SMLoc &EndLoc) {
                   .Case("alignto", AGVK::AGVK_AlignTo)
                   .Case("occupancy", AGVK::AGVK_Occupancy)
                   .Case("instprefsize", AGVK::AGVK_InstPrefSize)
+                  .Case("prefetchcachelines", AGVK::AGVK_PrefetchCachelines)
+                  .Case("prefetchoffset", AGVK::AGVK_PrefetchOffset)
                   .Default(AGVK::AGVK_None);
 
     if (VK != AGVK::AGVK_None && peekToken().is(AsmToken::LParen)) {
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp
index 5cd2aa86560ae..fbd15135da850 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp
@@ -91,6 +91,12 @@ static unsigned getFixupKindNumBytes(unsigned Kind) {
   switch (Kind) {
   case AMDGPU::fixup_si_sopp_br:
     return 2;
+  case AMDGPU::fixup_si_prefetch_sdata:
+    // The sdata field is 5 bits at bit offset 6, spanning bytes 0 and 1.
+    return 2;
+  case AMDGPU::fixup_si_prefetch_offset:
+    // The offset field is 24 bits at bit offset 32 (bytes 4-6 of 8-byte inst).
+    return 3;
   case FK_SecRel_1:
   case FK_Data_1:
     return 1;
@@ -121,6 +127,14 @@ static uint64_t adjustFixupValue(const MCFixup &Fixup, uint64_t Value,
 
     return BrImm;
   }
+  case AMDGPU::fixup_si_prefetch_sdata:
+    // The value is already the computed cacheline count from the MCExpr.
+    // Clamp to 5-bit field (max 31).
+    return std::min(Value, static_cast<uint64_t>(31));
+  case AMDGPU::fixup_si_prefetch_offset:
+    // The value is the byte offset. It's a 24-bit signed field.
+    // Clamp to valid range.
+    return SignedValue & 0xFFFFFF;
   case FK_Data_1:
   case FK_Data_2:
   case FK_Data_4:
@@ -179,6 +193,12 @@ MCFixupKindInfo AMDGPUAsmBackend::getFixupKindInfo(MCFixupKind Kind) const {
   const static MCFixupKindInfo Infos[AMDGPU::NumTargetFixupKinds] = {
       // name                   offset bits  flags
       {"fixup_si_sopp_br", 0, 16, 0},
+      // Prefetch sdata is a 5-bit field at bits 10-6 in the 64-bit SMEM
+      // instruction encoding (GFX12+).
+      {"fixup_si_prefetch_sdata", 6, 5, 0},
+      // Prefetch offset is a 24-bit field at bit 0 of the high 32-bits
+      // (byte offset 4 from instruction start).
+      {"fixup_si_prefetch_offset", 0, 24, 0},
   };
 
   if (mc::isRelocation(Kind))
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h
index d49bb196ab3a1..ff98f69c7507e 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h
@@ -17,6 +17,14 @@ enum Fixups {
   /// 16-bit PC relative fixup for SOPP branch instructions.
   fixup_si_sopp_br = FirstTargetFixupKind,
 
+  /// Fixups for s_prefetch_inst instructions.
+  /// These compute values based on code size at assembly time.
+  /// The sdata field (cacheline count) is a 5-bit field.
+  fixup_si_prefetch_sdata,
+
+  /// The offset field (byte offset) is a 24-bit signed field.
+  fixup_si_prefetch_offset,
+
   // Marker
   LastTargetFixupKind,
   NumTargetFixupKinds = LastTargetFixupKind - FirstTargetFixupKind
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.cpp
index 0aece53db1eb0..c9eed3f7158ee 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.cpp
@@ -145,7 +145,12 @@ void AMDGPUInstPrinter::printSMRDOffset8(const MCInst *MI, unsigned OpNo,
 void AMDGPUInstPrinter::printSMEMOffset(const MCInst *MI, unsigned OpNo,
                                         const MCSubtargetInfo &STI,
                                         raw_ostream &O) {
-  O << formatHex(MI->getOperand(OpNo).getImm());
+  const MCOperand &Op = MI->getOperand(OpNo);
+  if (Op.isExpr()) {
+    MAI.printExpr(O, *Op.getExpr());
+  } else {
+    O << formatHex(Op.getImm());
+  }
 }
 
 void AMDGPUInstPrinter::printSMRDLiteralOffset(const MCInst *MI, unsigned OpNo,
@@ -154,6 +159,17 @@ void AMDGPUInstPrinter::printSMRDLiteralOffset(const MCInst *MI, unsigned OpNo,
   printU32ImmOperand(MI, OpNo, STI, O);
 }
 
+void AMDGPUInstPrinter::printPrefetchSdata(const MCInst *MI, unsigned OpNo,
+                                           const MCSubtargetInfo &STI,
+                                           raw_ostream &O) {
+  const MCOperand &Op = MI->getOperand(OpNo);
+  if (Op.isExpr()) {
+    MAI.printExpr(O, *Op.getExpr());
+  } else {
+    O << Op.getImm();
+  }
+}
+
 void AMDGPUInstPrinter::printCPol(const MCInst *MI, unsigned OpNo,
                                   const MCSubtargetInfo &STI, raw_ostream &O) {
   auto Imm = MI->getOperand(OpNo).getImm();
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.h
index 5f5f15712b5ac..599070c70b428 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUInstPrinter.h
@@ -59,6 +59,8 @@ class AMDGPUInstPrinter : public MCInstPrinter {
                        const MCSubtargetInfo &STI, raw_ostream &O);
   void printSMRDLiteralOffset(const MCInst *MI, unsigned OpNo,
                               const MCSubtargetInfo &STI, raw_ostream &O);
+  void printPrefetchSdata(const MCInst *MI, unsigned OpNo,
+                          const MCSubtargetInfo &STI, raw_ostream &O);
   void printCPol(const MCInst *MI, unsigned OpNo,
                  const MCSubtargetInfo &STI, raw_ostream &O);
   void printTH(const MCInst *MI, int64_t TH, int64_t Scope, raw_ostream &O);
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCCodeEmitter.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCCodeEmitter.cpp
index 18c0919471d6d..328c02b285d60 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCCodeEmitter.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCCodeEmitter.cpp
@@ -513,7 +513,19 @@ void AMDGPUMCCodeEmitter::getSOPPBrEncoding(const MCInst &MI, unsigned OpNo,
 void AMDGPUMCCodeEmitter::getSMEMOffsetEncoding(
     const MCInst &MI, unsigned OpNo, APInt &Op,
     SmallVectorImpl<MCFixup> &Fixups, const MCSubtargetInfo &STI) const {
-  auto Offset = MI.getOperand(OpNo).getImm();
+  const MCOperand &MO = MI.getOperand(OpNo);
+  if (MO.isExpr()) {
+    const MCExpr *Expr = MO.getExpr();
+    if (Expr->getKind() == MCExpr::Target) {
+      const AMDGPUMCExpr *AExpr = static_cast<const AMDGPUMCExpr *>(Expr);
+      if (AExpr->getKind() == AMDGPUMCExpr::AGVK_PrefetchOffset) {
+        addFixup(Fixups, 4, Expr, AMDGPU::fixup_si_prefetch_offset);
+        Op = 0;
+        return;
+      }
+    }
+  }
+  auto Offset = MO.getImm();
   // VI only supports 20-bit unsigned offsets.
   assert(!AMDGPU::isVI(STI) || isUInt<20>(Offset));
   Op = Offset;
@@ -716,6 +728,29 @@ void AMDGPUMCCodeEmitter::getMachineOpValueCommon(
   } else if (MO.isExpr() && MO.getExpr()->evaluateAsAbsolute(Val)) {
     isLikeImm = true;
   } else if (MO.isExpr()) {
+    // Check for prefetch cacheline MCExpr - these need a special fixup
+    // because the expression computes a cacheline count based on code size.
+    const MCExpr *Expr = MO.getExpr();
+    if (Expr->getKind() == MCExpr::Target) {
+      const AMDGPUMCExpr *AExpr = static_cast<const AMDGPUMCExpr *>(Expr);
+      if (AExpr->getKind() == AMDGPUMCExpr::AGVK_PrefetchCachelines) {
+        // Create a fixup for the prefetch sdata field. The fixup will be
+        // applied after assembly when symbol positions are known.
+        // Offset 0 means from the start of the instruction.
+        addFixup(Fixups, 0, Expr, AMDGPU::fixup_si_prefetch_sdata);
+        // Set Op to 0 as placeholder; fixup will overwrite.
+        Op = 0;
+        return;
+      }
+      if (AExpr->getKind() == AMDGPUMCExpr::AGVK_PrefetchOffset) {
+        // Create a fixup for the prefetch offset field.
+        // Offset 4 means byte 4 of the 8-byte instruction.
+        addFixup(Fixups, 4, Expr, AMDGPU::fixup_si_prefetch_offset);
+        Op = 0;
+        return;
+      }
+    }
+
     // FIXME: If this is expression is PCRel or not should not depend on what
     // the expression looks like. Given that this is just a general expression,
     // it should probably be FK_Data_4 and whatever is producing
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index 745647e235f6e..17028fc9b26d9 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -68,6 +68,8 @@ unsigned AMDGPUMCExpr::getNumExpectedArgs(VariantKind Kind) {
     return 1;
   case AGVK_TotalNumVGPRs:
   case AGVK_AlignTo:
+  case AGVK_PrefetchCachelines:
+  case AGVK_PrefetchOffset:
     return 2;
   case AGVK_ExtraSGPRs:
     return 3;
@@ -110,6 +112,12 @@ void AMDGPUMCExpr::printImpl(raw_ostream &OS, const MCAsmInfo *MAI) const {
   case AGVK_InstPrefSize:
     OS << "instprefsize(";
     break;
+  case AGVK_PrefetchCachelines:
+    OS << "prefetchcachelines(";
+    break;
+  case AGVK_PrefetchOffset:
+    OS << "prefetchoffset(";
+    break;
   case AGVK_Lit:
     OS << "lit(";
     break;
@@ -245,6 +253,74 @@ bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
   return true;
 }
 
+bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
+                                              const MCAssembler *Asm) const {
+  uint64_t SlotIndex = 0, CodeSizeInBytes = 0;
+  if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes}))
+    return false;
+
+  // Constants for prefetch calculation.
+  // Each instruction can prefetch up to 31 cachelines (5-bit sdata field).
+  // Cacheline size is 128 bytes. Each slot covers ~4KB (31 * 128 = 3968 bytes).
+  constexpr unsigned MaxCachelinesPerPrefetch = 31;
+  constexpr unsigned CacheLineSize = 128;
+  constexpr unsigned BytesPerPrefetch =
+      MaxCachelinesPerPrefetch * CacheLineSize;
+  constexpr uint64_t MaxPrefetchSize = 64 * 1024; // 64KB ICache
+
+  // Clamp code size to maximum prefetchable size.
+  uint64_t PrefetchSize = std::min(CodeSizeInBytes, MaxPrefetchSize);
+
+  // Calculate the byte offset for this slot.
+  uint64_t SlotOffset = SlotIndex * BytesPerPrefetch;
+
+  // If this slot starts beyond the code size, return 0 (NOP).
+  if (SlotOffset >= PrefetchSize) {
+    Res = MCValue::get(static_cast<int64_t>(0));
+    return true;
+  }
+
+  // Calculate remaining bytes from this slot's offset.
+  uint64_t RemainingBytes = PrefetchSize - SlotOffset;
+
+  // Calculate cachelines needed, clamped to max per instruction.
+  uint64_t CachelinesNeeded = divideCeil(RemainingBytes, CacheLineSize);
+  uint64_t CachelineCount = std::min(
+      CachelinesNeeded, static_cast<uint64_t>(MaxCachelinesPerPrefetch));
+
+  Res = MCValue::get(static_cast<int64_t>(CachelineCount));
+  return true;
+}
+
+bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
+                                          const MCAssembler *Asm) const {
+  uint64_t SlotIndex = 0, CodeSizeInBytes = 0;
+  if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes}))
+    return false;
+
+  constexpr unsigned MaxCachelinesPerPrefetch = 31;
+  constexpr unsigned CacheLineSize = 128;
+  constexpr uint64_t MaxPrefetchSize = 64 * 1024;
+
+  uint64_t PrefetchSize = std::min(CodeSizeInBytes, MaxPrefetchSize);
+  uint64_t Offset = 0;
+  uint64_t Remaining = PrefetchSize;
+
+  for (uint64_t I = 0; I < SlotIndex; ++I) {
+    if (Remaining == 0)
+      break;
+    uint64_t Cachelines =
+        std::min(divideCeil(Remaining, CacheLineSize),
+                 static_cast<uint64_t>(MaxCachelinesPerPrefetch));
+    uint64_t PrefetchBytes = Cachelines * CacheLineSize;
+    Offset += PrefetchBytes;
+    Remaining = (Remaining > PrefetchBytes) ? Remaining - PrefetchBytes : 0;
+  }
+
+  Res = MCValue::get(static_cast<int64_t>(Offset));
+  return true;
+}
+
 bool AMDGPUMCExpr::isSymbolUsedInExpression(const MCSymbol *Sym,
                                             const MCExpr *E) {
   switch (E->getKind()) {
@@ -292,6 +368,10 @@ bool AMDGPUMCExpr::evaluateAsRelocatableImpl(MCValue &Res,
     return evaluateOccupancy(Res, Asm);
   case AGVK_InstPrefSize:
     return evaluateInstPrefSize(Res, Asm);
+  case AGVK_PrefetchCachelines:
+    return evaluatePrefetchCachelines(Res, Asm);
+  case AGVK_PrefetchOffset:
+    return evaluatePrefetchOffset(Res, Asm);
   case AGVK_Lit:
   case AGVK_Lit64:
     return Args[0]->evaluateAsRelocatable(Res, Asm);
@@ -349,6 +429,16 @@ AMDGPUMCExpr::createInstPrefSize(const MCExpr *CodeSizeBytes, MCContext &Ctx) {
   return create(AGVK_InstPrefSize, {CodeSizeBytes}, Ctx);
 }
 
+const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchCachelines(
+    const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes, MCContext &Ctx) {
+  return create(AGVK_PrefetchCachelines, {SlotIndex, CodeSizeBytes}, Ctx);
+}
+
+const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchOffset(
+    const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes, MCContext &Ctx) {
+  return create(AGVK_PrefetchOffset, {SlotIndex, CodeSizeBytes}, Ctx);
+}
+
 const AMDGPUMCExpr *AMDGPUMCExpr::createLit(LitModifier Lit, int64_t Value,
                                             MCContext &Ctx) {
   assert(Lit == LitModifier::Lit || Lit == LitModifier::Lit64);
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index 96003711faf88..a8d9385121b03 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -41,6 +41,8 @@ class AMDGPUMCExpr : public MCTargetExpr {
     AGVK_AlignTo,
     AGVK_Occupancy,
     AGVK_InstPrefSize,
+    AGVK_PrefetchCachelines,
+    AGVK_PrefetchOffset,
     AGVK_Lit,
     AGVK_Lit64,
     AGVK_Min,
@@ -74,6 +76,8 @@ class AMDGPUMCExpr : public MCTargetExpr {
   bool evaluateAlignTo(MCValue &Res, const MCAssembler *Asm) const;
   bool evaluateOccupancy(MCValue &Res, const MCAssembler *Asm) const;
   bool evaluateInstPrefSize(MCValue &Res, const MCAssembler *Asm) const;
+  bool evaluatePrefetchCachelines(MCValue &Res, const MCAssembler *Asm) const;
+  bool evaluatePrefetchOffset(MCValue &Res, const MCAssembler *Asm) const;
 
 public:
   static const AMDGPUMCExpr *
@@ -116,6 +120,23 @@ class AMDGPUMCExpr : public MCTargetExpr {
   static const AMDGPUMCExpr *createInstPrefSize(const MCExpr *CodeSizeBytes,
                                                 MCContext &Ctx);
 
+  /// Create an expression for computing cacheline count for a prefetch slot.
+  /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
+  /// CodeSizeBytes is the total code size in bytes.
+  /// Returns the number of cachelines this slot should prefetch.
+  static const AMDGPUMCExpr *
+  createPrefetchCachelines(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
+                           MCContext &Ctx);
+
+  /// Create an expression for computing the byte offset for a prefetch slot.
+  /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
+  /// CodeSizeBytes is the total code size in bytes.
+  /// Returns the cumulative byte offset where this slot should start
+  /// prefetching.
+  static const AMDGPUMCExpr *createPrefetchOffset(const MCExpr *SlotIndex,
+                                                  const MCExpr *CodeSizeBytes,
+                                                  MCContext &Ctx);
+
   static const AMDGPUMCExpr *createLit(LitModifier Lit, int64_t Value,
                                        MCContext &Ctx);
 
diff --git a/llvm/lib/Target/AMDGPU/SIDefines.h b/llvm/lib/Target/AMDGPU/SIDefines.h
index 864a84cc8daf4..a67c259541378 100644
--- a/llvm/lib/Target/AMDGPU/SIDefines.h
+++ b/llvm/lib/Target/AMDGPU/SIDefines.h
@@ -824,6 +824,9 @@ enum ModeRegisterMasks : uint32_t {
   SRC2_VGPR_MSB = 0x3 << 18,
   VGPR_MSB_MASK = 0xff << 12, // Bits 12..19
 
+  // GFX1250 instruction prefetch
+  SCALAR_PREFETCH_EN = 1 << 24,
+
   REPLAY_MODE = 1 << 25,
   FLAT_SCRATCH_IS_NV = 1 << 26,
 };
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
index 0207c728ea9b8..4b85168bdfed6 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
@@ -490,6 +490,10 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
   bool HasNonSpillStackObjects = false;
   bool IsStackRealigned = false;
 
+  // Set when ICache prefetch instructions have been inserted in the entry
+  // block. This tells AsmPrinter to set rsrc3 INST_PREF_SIZE to 0.
+  bool HasICachePrefetch = false;
+
   unsigned NumSpilledSGPRs = 0;
   unsigned NumSpilledVGPRs = 0;
 
@@ -1126,6 +1130,12 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
     IsStackRealigned = Realigned;
   }
 
+  bool hasICachePrefetch() const { return HasICachePrefetch; }
+
+  void setHasICachePrefetch(bool Prefetch = true) {
+    HasICachePrefetch = Prefetch;
+  }
+
   unsigned getNumSpilledSGPRs() const {
     return NumSpilledSGPRs;
   }
diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index 9b67cdd6be6f4..608f03b1dedce 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -20,12 +20,17 @@
 
 #include "AMDGPU.h"
 #include "GCNSubtarget.h"
+#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
+#include "SIMachineFunctionInfo.h"
 #include "llvm/ADT/Statistic.h"
 #include "llvm/CodeGen/MachineDominators.h"
 #include "llvm/CodeGen/MachineFunctionPass.h"
+#include "llvm/CodeGen/MachineInstrBuilder.h"
 #include "llvm/CodeGen/MachineLoopInfo.h"
 #include "llvm/CodeGen/TargetSchedule.h"
 #include "llvm/Support/BranchProbability.h"
+#include "llvm/Support/CommandLine.h"
+#include "llvm/TargetParser/Triple.h"
 using namespace llvm;
 
 #define DEBUG_TYPE "si-pre-emit-peephole"
@@ -45,12 +50,26 @@ struct ModeFieldState {
   bool isTracked() const { return PendingWrite || Value; }
 };
 
+static cl::opt<bool>
+    EnableICachePrefetch("amdgpu-icache-prefetch",
+                         cl::desc("Insert ICache prefetch instructions"),
+                         cl::init(true), cl::Hidden);
+
+namespace {
+
+// Number of prefetch instructions to insert.
+// Each can prefetch up to 31 cachelines of 128 bytes = ~4KB.
+// 16 instructions cover 64KB (the full ICache size).
+static constexpr unsigned NumPrefetchInsts = 16;
+
 class SIPreEmitPeephole {
 private:
+  const GCNSubtarget *ST = nullptr;
   const SIInstrInfo *TII = nullptr;
   const SIRegisterInfo *TRI = nullptr;
   MachineLoopInfo *MLI = nullptr;
 
+  bool insertICachePrefetch(MachineFunction &MF);
   bool optimizeVccBranch(MachineInstr &MI) const;
   void updateMLIBeforeRemovingEdge(MachineBasicBlock *From,
                                    MachineBasicBlock *To) const;
@@ -191,8 +210,7 @@ bool SIPreEmitPeephole::optimizeVccBranch(MachineInstr &MI) const {
 
   bool Changed = false;
   MachineBasicBlock &MBB = *MI.getParent();
-  const GCNSubtarget &ST = MBB.getParent()->getSubtarget<GCNSubtarget>();
-  const bool IsWave32 = ST.isWave32();
+  const bool IsWave32 = ST->isWave32();
   const unsigned CondReg = TRI->getVCC();
   const unsigned ExecReg = IsWave32 ? AMDGPU::EXEC_LO : AMDGPU::EXEC;
   const unsigned And = IsWave32 ? AMDGPU::S_AND_B32 : AMDGPU::S_AND_B64;
@@ -850,6 +868,73 @@ MachineInstrBuilder SIPreEmitPeephole::createUnpackedMI(MachineInstr &I,
   return NewMI;
 }
 
+bool SIPreEmitPeephole::insertICachePrefetch(MachineFunction &MF) {
+  // Only run on targets that support ICache prefetching.
+  if (!ST->hasICachePrefetch())
+    return false;
+
+  // Only run for AMDHSA - this is where kernel descriptors are used
+  // and rsrc3 INST_PREF_SIZE is relevant.
+  const Triple &TT = ST->getTargetTriple();
+  if (TT.getOS() != Triple::AMDHSA)
+    return false;
+
+  SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
+
+  // Only insert prefetch instructions for entry functions.
+  if (!MFI->isEntryFunction())
+    return false;
+
+  MachineBasicBlock &EntryBB = MF.front();
+  MachineBasicBlock::iterator InsertPt = EntryBB.begin();
+
+  // Skip past any instructions that must remain at the very beginning:
+  // - Debug values and CFI instructions
+  // - S_SETREG_IMM32_B32 instructions that set up MODE register bits
+  //   (e.g., REPLAY_MODE bit 25 from SIFrameLowering)
+  // We want the prefetches to come after all initial MODE setup.
+  while (InsertPt != EntryBB.end()) {
+    if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction()) {
+      ++InsertPt;
+      continue;
+    }
+    if (InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32) {
+      ++InsertPt;
+      continue;
+    }
+    break;
+  }
+
+  DebugLoc DL;
+
+  // Insert s_setreg_imm32_b32 to set MODE.SCALAR_PREFETCH_EN (bit 24).
+  // This ensures only the first wave in the WGP executes the prefetches.
+  // Insert it right before the prefetch instructions (after other MODE setup).
+  using namespace AMDGPU::Hwreg;
+  unsigned ModeRegEncoding = HwregEncoding::encode(ID_MODE, 24, 1);
+  BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_SETREG_IMM32_B32))
+      .addImm(1) // Value to set (enable)
+      .addImm(ModeRegEncoding);
+
+  // Insert 16 s_prefetch_inst_pc_rel instructions.
+  // The offset and sdata operands are placeholders - the sdata operand stores
+  // the slot index (0-15). Both will be fixed up in AMDGPUAsmPrinter based on
+  // the actual code size.
+  for (unsigned I = 0; I < NumPrefetchInsts; ++I) {
+    BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
+        .addImm(0)                 // offset (placeholder, fixed up later)
+        .addReg(AMDGPU::SGPR_NULL) // soffset
+        .addImm(I);                // sdata (slot index, fixed up later)
+  }
+
+  // Mark that we've inserted ICache prefetch instructions.
+  // This tells AsmPrinter to set rsrc3 INST_PREF_SIZE to 0 and fix up
+  // the prefetch cacheline counts.
+  MFI->setHasICachePrefetch(true);
+
+  return true;
+}
+
 PreservedAnalyses
 llvm::SIPreEmitPeepholePass::run(MachineFunction &MF,
                                  MachineFunctionAnalysisManager &MFAM) {
@@ -866,12 +951,16 @@ llvm::SIPreEmitPeepholePass::run(MachineFunction &MF,
 }
 
 bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
-  const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
-  TII = ST.getInstrInfo();
+  ST = &MF.getSubtarget<GCNSubtarget>();
+  TII = ST->getInstrInfo();
   TRI = &TII->getRegisterInfo();
   MLI = LoopInfo;
   bool Changed = false;
 
+  // Insert ICache prefetch instructions if enabled.
+  if (EnableICachePrefetch)
+    Changed |= insertICachePrefetch(MF);
+
   MF.RenumberBlocks();
 
   for (MachineBasicBlock &MBB : MF) {
@@ -892,7 +981,7 @@ bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
       }
     }
 
-    if (!ST.hasVGPRIndexMode())
+    if (!ST->hasVGPRIndexMode())
       continue;
 
     MachineInstr *SetGPRMI = nullptr;
@@ -929,7 +1018,7 @@ bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
   // side effects.
 
   // Perform the extra MF scans only for supported archs
-  if (!ST.hasGFX940Insts())
+  if (!ST->hasGFX940Insts())
     return Changed;
   for (MachineBasicBlock &MBB : MF) {
     // Unpack packed instructions overlapped by MFMAs. This allows the
diff --git a/llvm/lib/Target/AMDGPU/SMInstructions.td b/llvm/lib/Target/AMDGPU/SMInstructions.td
index 19aeafe9b30cc..7f75c77b9ea0c 100644
--- a/llvm/lib/Target/AMDGPU/SMInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SMInstructions.td
@@ -8,6 +8,9 @@
 
 def smrd_offset_8 : ImmOperand<i32, "SMRDOffset8", 1>;
 
+// Prefetch sdata operand - accepts immediates or MCExpr for ICache prefetch.
+def PrefetchSdata : ImmOperand<i8, "PrefetchSdata", 1>;
+
 let EncoderMethod = "getSMEMOffsetEncoding",
     DecoderMethod = "decodeSMEMOffset" in {
 def SMEMOffset : ImmOperand<i32, "SMEMOffset", 1>;
@@ -239,7 +242,7 @@ class SM_WaveId_Pseudo<string opName, SDPatternOperator node> : SM_Pseudo<
 
 class SM_Prefetch_Pseudo <string opName, RegisterClass baseClass, bit hasSBase>
   : SM_Pseudo<opName, (outs), !con(!if(hasSBase, (ins baseClass:$sbase), (ins)),
-                                   (ins SMEMOffset:$offset, SReg_32:$soffset, i8imm:$sdata, CPol_0:$cpol)),
+                                   (ins SMEMOffset:$offset, SReg_32:$soffset, PrefetchSdata:$sdata, CPol_0:$cpol)),
               !if(hasSBase, " $sbase,", "") # " $offset, $soffset, $sdata$cpol"> {
   // Mark prefetches as both load and store to prevent reordering with loads
   // and stores. This is also needed for pattern to match prefetch intrinsic.
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
new file mode 100644
index 0000000000000..2dc738cd703a6
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -0,0 +1,112 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -o - %s | \
+; RUN:   FileCheck -check-prefix=GFX1250 %s
+
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=false -o - %s | \
+; RUN:   FileCheck -check-prefix=NO-PREFETCH %s
+
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
+; RUN: llvm-objdump -d %t.o | FileCheck -check-prefix=GFX1250-OBJ %s
+
+; Test that ICache prefetch instructions are inserted for entry functions on gfx1250.
+; Verify object file has resolved cacheline counts.
+; GFX1250-OBJ-LABEL: <entry_kernel>:
+; GFX1250-OBJ:       s_prefetch_inst_pc_rel 0x0, null, 2
+; GFX1250-OBJ-NEXT:  s_prefetch_inst_pc_rel 0x100, null, 0
+
+define amdgpu_kernel void @entry_kernel(ptr addrspace(1) %out) {
+; GFX1250-LABEL: entry_kernel:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 24, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(0, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(1, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(2, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(3, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(4, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(5, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(6, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(7, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(8, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(9, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(10, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(11, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(12, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(13, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(14, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(15, .Lpref_func_end0-entry_kernel)
+; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
+; GFX1250-NEXT:    v_mov_b32_e32 v0, 0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    global_store_b32 v0, v0, s[0:1]
+; GFX1250-NEXT:    s_endpgm
+; GFX1250-NEXT:  .Lpref_func_end0:
+;
+; NO-PREFETCH-LABEL: entry_kernel:
+; NO-PREFETCH:       ; %bb.0:
+; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
+; NO-PREFETCH-NEXT:    v_mov_b32_e32 v0, 0
+; NO-PREFETCH-NEXT:    s_wait_kmcnt 0x0
+; NO-PREFETCH-NEXT:    global_store_b32 v0, v0, s[0:1]
+; NO-PREFETCH-NEXT:    s_endpgm
+  store i32 0, ptr addrspace(1) %out
+  ret void
+}
+
+; Verify that called functions do NOT get prefetch instructions.
+
+define void @called_function(ptr addrspace(1) %out) {
+; GFX1250-LABEL: called_function:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_mov_b32_e32 v2, 1
+; GFX1250-NEXT:    global_store_b32 v[0:1], v2, off
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+;
+; NO-PREFETCH-LABEL: called_function:
+; NO-PREFETCH:       ; %bb.0:
+; NO-PREFETCH-NEXT:    s_wait_loadcnt_dscnt 0x0
+; NO-PREFETCH-NEXT:    s_wait_kmcnt 0x0
+; NO-PREFETCH-NEXT:    v_mov_b32_e32 v2, 1
+; NO-PREFETCH-NEXT:    global_store_b32 v[0:1], v2, off
+; NO-PREFETCH-NEXT:    s_set_pc_i64 s[30:31]
+  store i32 1, ptr addrspace(1) %out
+  ret void
+}
+
+; GFX1250-OBJ-LABEL: <tiny_kernel>:
+; GFX1250-OBJ:       s_prefetch_inst_pc_rel 0x0, null, 2
+; GFX1250-OBJ-NEXT:  s_prefetch_inst_pc_rel 0x100, null, 0
+
+define amdgpu_kernel void @tiny_kernel() {
+; GFX1250-LABEL: tiny_kernel:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 24, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(0, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(1, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(2, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(3, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(4, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(5, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(6, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(7, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(8, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(9, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(10, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(11, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(12, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(13, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(14, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(15, .Lpref_func_end1-tiny_kernel)
+; GFX1250-NEXT:    s_endpgm
+; GFX1250-NEXT:  .Lpref_func_end1:
+;
+; NO-PREFETCH-LABEL: tiny_kernel:
+; NO-PREFETCH:       ; %bb.0:
+; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT:    s_endpgm
+  ret void
+}

>From 1d4a3033cc0b28997b9f8f51c8c2b3968ac71dc8 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 30 Jul 2026 11:35:12 -0500
Subject: [PATCH 02/22] Account for sdata increment

Instruction semantics for `s_prefetch_inst` adds 1 to the encoded
`sdata` for the number of cache lines to prefetch. Account for that
increment in the calculations.

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 .../Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp   |  4 ++--
 .../lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h |  2 +-
 llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp  | 11 ++++++-----
 llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h    |  6 ++++--
 llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp          |  6 +++---
 5 files changed, 16 insertions(+), 13 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp
index fbd15135da850..1acf18b757e0f 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUAsmBackend.cpp
@@ -128,8 +128,8 @@ static uint64_t adjustFixupValue(const MCFixup &Fixup, uint64_t Value,
     return BrImm;
   }
   case AMDGPU::fixup_si_prefetch_sdata:
-    // The value is already the computed cacheline count from the MCExpr.
-    // Clamp to 5-bit field (max 31).
+    // The value is already the encoded sdata field value from the MCExpr.
+    // Clamp to the maximum 5-bit encoded field value (31).
     return std::min(Value, static_cast<uint64_t>(31));
   case AMDGPU::fixup_si_prefetch_offset:
     // The value is the byte offset. It's a 24-bit signed field.
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h
index ff98f69c7507e..caa3930411a9b 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUFixupKinds.h
@@ -19,7 +19,7 @@ enum Fixups {
 
   /// Fixups for s_prefetch_inst instructions.
   /// These compute values based on code size at assembly time.
-  /// The sdata field (cacheline count) is a 5-bit field.
+  /// The encoded 5-bit sdata field (cacheline count minus one).
   fixup_si_prefetch_sdata,
 
   /// The offset field (byte offset) is a 24-bit signed field.
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index 17028fc9b26d9..9459f3e32ac1c 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -260,9 +260,9 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
     return false;
 
   // Constants for prefetch calculation.
-  // Each instruction can prefetch up to 31 cachelines (5-bit sdata field).
-  // Cacheline size is 128 bytes. Each slot covers ~4KB (31 * 128 = 3968 bytes).
-  constexpr unsigned MaxCachelinesPerPrefetch = 31;
+  // Each instruction can prefetch up to 32 cachelines (5-bit sdata field, plus
+  // one added). Cacheline size is 128 bytes. Each slot covers 4KiB.
+  constexpr unsigned MaxCachelinesPerPrefetch = 32;
   constexpr unsigned CacheLineSize = 128;
   constexpr unsigned BytesPerPrefetch =
       MaxCachelinesPerPrefetch * CacheLineSize;
@@ -288,7 +288,8 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
   uint64_t CachelineCount = std::min(
       CachelinesNeeded, static_cast<uint64_t>(MaxCachelinesPerPrefetch));
 
-  Res = MCValue::get(static_cast<int64_t>(CachelineCount));
+  // The instruction adds 1 to the encoded sdata, so deduct it here.
+  Res = MCValue::get(static_cast<int64_t>(CachelineCount - 1));
   return true;
 }
 
@@ -298,7 +299,7 @@ bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
   if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes}))
     return false;
 
-  constexpr unsigned MaxCachelinesPerPrefetch = 31;
+  constexpr unsigned MaxCachelinesPerPrefetch = 32;
   constexpr unsigned CacheLineSize = 128;
   constexpr uint64_t MaxPrefetchSize = 64 * 1024;
 
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index a8d9385121b03..d3693fb14045e 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -120,10 +120,12 @@ class AMDGPUMCExpr : public MCTargetExpr {
   static const AMDGPUMCExpr *createInstPrefSize(const MCExpr *CodeSizeBytes,
                                                 MCContext &Ctx);
 
-  /// Create an expression for computing cacheline count for a prefetch slot.
+  /// Create an expression for computing the encoded sdata field for a prefetch
+  /// slot.
   /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
   /// CodeSizeBytes is the total code size in bytes.
-  /// Returns the number of cachelines this slot should prefetch.
+  /// Returns the requested cacheline count minus one, encoded for the 5-bit
+  /// sdata field.
   static const AMDGPUMCExpr *
   createPrefetchCachelines(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
                            MCContext &Ctx);
diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index 608f03b1dedce..23745b575158e 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -58,9 +58,9 @@ static cl::opt<bool>
 namespace {
 
 // Number of prefetch instructions to insert.
-// Each can prefetch up to 31 cachelines of 128 bytes = ~4KB.
-// 16 instructions cover 64KB (the full ICache size).
-static constexpr unsigned NumPrefetchInsts = 16;
+// Each can prefetch up to 32 cachelines of 128 bytes = 4KiB.
+// 16 instructions cover 64KiB (the full ICache size).
+static constexpr unsigned MaxNumPrefetchInsts = 16;
 
 class SIPreEmitPeephole {
 private:

>From aa014988acaf08c501fa18e8b672a408407cac31 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 30 Jul 2026 11:41:09 -0500
Subject: [PATCH 03/22] Use code size estimate for prefetch insertion

When inserting explicit instruction cache prefetch instructions, take
the estimated code size into account:
- If the program is less than 32KiB, use the previous mechanism in the
  kernel descriptor instead of explicit instructions.
- Limit the number of prefetch instructions inserted to the necessary
  number plus some slack.

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp | 47 ++++++++++++--------
 1 file changed, 28 insertions(+), 19 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index 23745b575158e..b5bbb6dd23f4b 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -22,6 +22,7 @@
 #include "GCNSubtarget.h"
 #include "MCTargetDesc/AMDGPUMCTargetDesc.h"
 #include "SIMachineFunctionInfo.h"
+#include "SIProgramInfo.h"
 #include "llvm/ADT/Statistic.h"
 #include "llvm/CodeGen/MachineDominators.h"
 #include "llvm/CodeGen/MachineFunctionPass.h"
@@ -57,11 +58,6 @@ static cl::opt<bool>
 
 namespace {
 
-// Number of prefetch instructions to insert.
-// Each can prefetch up to 32 cachelines of 128 bytes = 4KiB.
-// 16 instructions cover 64KiB (the full ICache size).
-static constexpr unsigned MaxNumPrefetchInsts = 16;
-
 class SIPreEmitPeephole {
 private:
   const GCNSubtarget *ST = nullptr;
@@ -885,6 +881,16 @@ bool SIPreEmitPeephole::insertICachePrefetch(MachineFunction &MF) {
   if (!MFI->isEntryFunction())
     return false;
 
+  SIProgramInfo PI;
+  uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
+  // The kernel descriptor can specify an instruction prefetch size of up to 256
+  // in INST_PREF_SIZE. At a granularity of 128B, this equals 32KiB of
+  // instructions that can be prefetched without inserting explicit prefetch
+  // instructions.
+  constexpr uint64_t MaxKDPrefetch = 1u << 15;
+  if (ProgramSize <= MaxKDPrefetch)
+    return false;
+
   MachineBasicBlock &EntryBB = MF.front();
   MachineBasicBlock::iterator InsertPt = EntryBB.begin();
 
@@ -907,20 +913,23 @@ bool SIPreEmitPeephole::insertICachePrefetch(MachineFunction &MF) {
 
   DebugLoc DL;
 
-  // Insert s_setreg_imm32_b32 to set MODE.SCALAR_PREFETCH_EN (bit 24).
-  // This ensures only the first wave in the WGP executes the prefetches.
-  // Insert it right before the prefetch instructions (after other MODE setup).
-  using namespace AMDGPU::Hwreg;
-  unsigned ModeRegEncoding = HwregEncoding::encode(ID_MODE, 24, 1);
-  BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_SETREG_IMM32_B32))
-      .addImm(1) // Value to set (enable)
-      .addImm(ModeRegEncoding);
-
-  // Insert 16 s_prefetch_inst_pc_rel instructions.
-  // The offset and sdata operands are placeholders - the sdata operand stores
-  // the slot index (0-15). Both will be fixed up in AMDGPUAsmPrinter based on
-  // the actual code size.
-  for (unsigned I = 0; I < NumPrefetchInsts; ++I) {
+  // Calculate the number of prefetch instructions required for the current
+  // program size. Each prefetch can transfer 4KiB of instructions. Add
+  // some slack as inserting the prefetches and later transformations, e.g.,
+  // padding and alignment, will introduce additional bytes. The offset and
+  // sdata operands are placeholders - the sdata operand stores the slot index
+  // (0-15). Both will be fixed up in AMDGPUAsmPrinter based on the actual code
+  // size.
+  constexpr uint64_t PrefetchSlack = 2 * 1024;
+  constexpr uint64_t BytesPerPrefetch = 4 * 1024;
+  // Each prefetch can transfer up to 32 cachelines of 128 bytes = 4KiB.
+  // 16 instructions cover 64KiB (the full ICache size).
+  constexpr unsigned MaxNumPrefetchInsts = 16;
+  unsigned NumPrefetches =
+      llvm::divideCeil(ProgramSize + PrefetchSlack, BytesPerPrefetch);
+  // Limit to 16 prefetches at most, otherwise we'd exceed the cache size.
+  NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
+  for (unsigned I = 0; I < NumPrefetches; ++I) {
     BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
         .addImm(0)                 // offset (placeholder, fixed up later)
         .addReg(AMDGPU::SGPR_NULL) // soffset

>From 74db290de8208f0d8c45276454c90143031eba3a Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 31 Jul 2026 03:33:29 -0500
Subject: [PATCH 04/22] Set INST_PREF_SIZE to 1 for explicit prefetch

Setting INST_PREF_SIZE to 1 is safe and activates the prefetching
feature. This means the first wave on the WGP will have its
SCALAR_PREFETCH_EN set to 1, while it is deactivated for the other
waves.

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp  | 10 ++++++----
 llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp |  2 +-
 2 files changed, 7 insertions(+), 5 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 7ca9ebd5db622..e3bf41f8e513f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -264,16 +264,18 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
   // right after the function code, so (Lfunc_end - func_sym) gives the
   // exact function code size in bytes.
   //
-  // When ICache prefetch is enabled, we set INST_PREF_SIZE to 0 because
-  // the s_prefetch_inst instructions handle all prefetching.
+  // When ICache prefetch is enabled, we set INST_PREF_SIZE to 1 because
+  // the s_prefetch_inst instructions handle the actual prefetching.
   if (STM.hasInstPrefSize()) {
     uint32_t Mask, Shift, Width, CacheLineSize;
     STM.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
 
     const MCExpr *InstPrefSize;
     if (MFI.hasICachePrefetch()) {
-      // Disable hardware prefetch - s_prefetch_inst handles it.
-      InstPrefSize = MCConstantExpr::create(0, Ctx);
+      // Set INST_PREF_SIZE to 1. This enables prefetch and the first wave on
+      // the WGP gets SCALAR_PREFETCH_EN set to 1, while the remaining waves
+      // receive 0. This enables the s_prefetch_inst for the first wave.
+      InstPrefSize = MCConstantExpr::create(1, Ctx);
     } else {
       const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
           MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index b5bbb6dd23f4b..11344ff8e4709 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -937,7 +937,7 @@ bool SIPreEmitPeephole::insertICachePrefetch(MachineFunction &MF) {
   }
 
   // Mark that we've inserted ICache prefetch instructions.
-  // This tells AsmPrinter to set rsrc3 INST_PREF_SIZE to 0 and fix up
+  // This tells AsmPrinter to set rsrc3 INST_PREF_SIZE to 1 and fix up
   // the prefetch cacheline counts.
   MFI->setHasICachePrefetch(true);
 

>From 0cb3b7013ec61921f9a8feb5d1c768999da82bc6 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 31 Jul 2026 08:24:19 -0500
Subject: [PATCH 05/22] Account for PC in offset calculcation

The `s_prefetch_inst_pc_rel` instruction prefetches relative to its own
PC. Account for that in the offset calculation to avoid gaps in the
prefetched data.

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp   | 12 ++-
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h     |  6 ++
 llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp  | 17 +++-
 .../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp      | 96 ++++++++++++-------
 .../Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h | 16 ++--
 5 files changed, 100 insertions(+), 47 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index e3bf41f8e513f..25cd2516e145c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -193,10 +193,13 @@ void AMDGPUAsmPrinter::emitFunctionBodyStart() {
   const GCNSubtarget &STM = MF->getSubtarget<GCNSubtarget>();
   const Function &F = MF->getFunction();
 
-  // If ICache prefetch is enabled, create the function end symbol early so
-  // it can be referenced by the prefetch MCExprs during instruction emission.
-  if (MFI.hasICachePrefetch())
+  // If ICache prefetch is enabled, create the function end symbol and prefetch
+  // block start symbol early so it can be referenced by the prefetch MCExprs
+  // during instruction emission.
+  if (MFI.hasICachePrefetch()) {
     PrefetchEndSym = createTempSymbol("pref_func_end");
+    PrefetchBlockStartSym = createTempSymbol("pref_block_start");
+  }
 
   // TODO: We're checking this late, would be nice to check it earlier.
   if (STM.requiresCodeObjectV6() && CodeObjectVersion < AMDGPU::AMDHSA_COV6) {
@@ -230,8 +233,9 @@ void AMDGPUAsmPrinter::emitFunctionBodyEnd() {
   // This symbol was created in emitFunctionBodyStart for this function.
   if (PrefetchEndSym) {
     OutStreamer->emitLabel(PrefetchEndSym);
-    PrefetchEndSym = nullptr;
   }
+  PrefetchEndSym = nullptr;
+  PrefetchBlockStartSym = nullptr;
 }
 
 /// Set bits in a kernel descriptor MCExpr field:
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
index 871bd738b947b..22eb7da72812b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
@@ -60,6 +60,9 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
   // Symbol for the function end, used by ICache prefetch MCExprs.
   // Created early in emitFunctionBodyStart when prefetch is enabled.
   MCSymbol *PrefetchEndSym = nullptr;
+  // Symbol for the first prefetch instruction, used by ICache prefetch MCExprs.
+  // Created early in emitFunctionBodyStart when prefetch is enabled.
+  MCSymbol *PrefetchBlockStartSym = nullptr;
 
   // When appropriate, add a _dvgpr$ symbol.
   void emitDVgprSymbol(MachineFunction &MF);
@@ -164,6 +167,9 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
 
   /// Get the symbol for the function end, used for ICache prefetch MCExprs.
   MCSymbol *getPrefetchEndSym() const { return PrefetchEndSym; }
+  /// Get the symbol for the first prefetch instruction, used for ICache
+  /// prefetch MCExprs.
+  MCSymbol *getPrefetchBlockStartSym() const { return PrefetchBlockStartSym; }
 
   /// Get the code size estimate from SIProgramInfo.
   uint64_t getCodeSize() { return CurrentProgramInfo.getFunctionCodeSize(*MF); }
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
index fd4651a41f8a8..be97f392d9bf4 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
@@ -477,9 +477,14 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
         constexpr unsigned OffsetIdx = 0;
         constexpr unsigned SdataIdx = 2;
 
-        // The sdata operand contains the slot index (0-15) set by the pass.
+        // The sdata operand contains the slot index [0, N) set by the pass.
         int64_t SlotIndex = TmpInst.getOperand(SdataIdx).getImm();
 
+        // If this is the first slot, emit the prefetch block start symbol
+        // before the instruction.
+        if (SlotIndex == 0)
+          OutStreamer->emitLabel(getPrefetchBlockStartSym());
+
         // Create MCExpr for code size using label subtraction.
         // This gives the exact code size at assembly time.
         const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
@@ -490,12 +495,18 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
         const MCExpr *SlotIndexExpr =
             MCConstantExpr::create(SlotIndex, OutContext);
 
+        // Create MCExpr for the offset of the first prefetch instruction in the
+        // function.
+        const MCExpr *PrefetchBlockOffset = MCBinaryExpr::createSub(
+            MCSymbolRefExpr::create(getPrefetchBlockStartSym(), OutContext),
+            MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
+
         // Create MCExprs that will be evaluated at fixup time when symbol
         // positions are known.
         const MCExpr *CachelinesExpr = AMDGPUMCExpr::createPrefetchCachelines(
-            SlotIndexExpr, CodeSizeExpr, OutContext);
+            SlotIndexExpr, CodeSizeExpr, PrefetchBlockOffset, OutContext);
         const MCExpr *OffsetExpr = AMDGPUMCExpr::createPrefetchOffset(
-            SlotIndexExpr, CodeSizeExpr, OutContext);
+            SlotIndexExpr, CodeSizeExpr, PrefetchBlockOffset, OutContext);
 
         // Replace the offset and sdata operands with MCExprs.
         TmpInst.getOperand(OffsetIdx) = MCOperand::createExpr(OffsetExpr);
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index 9459f3e32ac1c..d349a2f29af11 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -68,10 +68,10 @@ unsigned AMDGPUMCExpr::getNumExpectedArgs(VariantKind Kind) {
     return 1;
   case AGVK_TotalNumVGPRs:
   case AGVK_AlignTo:
-  case AGVK_PrefetchCachelines:
-  case AGVK_PrefetchOffset:
     return 2;
   case AGVK_ExtraSGPRs:
+  case AGVK_PrefetchCachelines:
+  case AGVK_PrefetchOffset:
     return 3;
   case AGVK_Occupancy:
     return 9;
@@ -253,10 +253,34 @@ bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
   return true;
 }
 
+static uint64_t calcFirstPrefetchOffset(uint64_t PrefetchBlockOffset) {
+  constexpr unsigned CacheLineSize = 128;
+  // Start prefetching on the first cache line after the first prefetch
+  // instruction.
+  return llvm::alignDown(PrefetchBlockOffset, CacheLineSize) + CacheLineSize;
+}
+
+static uint64_t calcPrefetchSize(uint64_t CodeSizeInBytes,
+                                 uint64_t PrefetchBlockOffset) {
+  constexpr uint64_t MaxPrefetchSize = 64 * 1024; // 64KB ICache
+  uint64_t FirstPrefetch = calcFirstPrefetchOffset(PrefetchBlockOffset);
+  uint64_t ClampedCodeSize = std::min(CodeSizeInBytes, MaxPrefetchSize);
+  return ClampedCodeSize > FirstPrefetch ? ClampedCodeSize - FirstPrefetch : 0;
+}
+
+static uint64_t calcPrefetchSlotOffset(uint64_t SlotIndex) {
+  constexpr unsigned MaxCachelinesPerPrefetch = 32;
+  constexpr unsigned CacheLineSize = 128;
+  constexpr unsigned BytesPerPrefetch =
+      MaxCachelinesPerPrefetch * CacheLineSize;
+  return SlotIndex * BytesPerPrefetch;
+}
+
 bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
                                               const MCAssembler *Asm) const {
-  uint64_t SlotIndex = 0, CodeSizeInBytes = 0;
-  if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes}))
+  uint64_t SlotIndex = 0, CodeSizeInBytes = 0, PrefetchBlockOffset = 0;
+  if (!evaluateMCExprs(Args, Asm,
+                       {SlotIndex, CodeSizeInBytes, PrefetchBlockOffset}))
     return false;
 
   // Constants for prefetch calculation.
@@ -264,17 +288,15 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
   // one added). Cacheline size is 128 bytes. Each slot covers 4KiB.
   constexpr unsigned MaxCachelinesPerPrefetch = 32;
   constexpr unsigned CacheLineSize = 128;
-  constexpr unsigned BytesPerPrefetch =
-      MaxCachelinesPerPrefetch * CacheLineSize;
-  constexpr uint64_t MaxPrefetchSize = 64 * 1024; // 64KB ICache
-
-  // Clamp code size to maximum prefetchable size.
-  uint64_t PrefetchSize = std::min(CodeSizeInBytes, MaxPrefetchSize);
+  uint64_t PrefetchSize =
+      calcPrefetchSize(CodeSizeInBytes, PrefetchBlockOffset);
 
   // Calculate the byte offset for this slot.
-  uint64_t SlotOffset = SlotIndex * BytesPerPrefetch;
+  uint64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
 
-  // If this slot starts beyond the code size, return 0 (NOP).
+  // If this slot starts beyond the prefetchable region, use the minimum
+  // encoded prefetch size. evaluatePrefetchOffset() targets the instruction's
+  // own cache line for such slots.
   if (SlotOffset >= PrefetchSize) {
     Res = MCValue::get(static_cast<int64_t>(0));
     return true;
@@ -295,29 +317,31 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
 
 bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
                                           const MCAssembler *Asm) const {
-  uint64_t SlotIndex = 0, CodeSizeInBytes = 0;
-  if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes}))
+  uint64_t SlotIndex = 0, CodeSizeInBytes = 0, PrefetchBlockOffset = 0;
+  if (!evaluateMCExprs(Args, Asm,
+                       {SlotIndex, CodeSizeInBytes, PrefetchBlockOffset}))
     return false;
 
-  constexpr unsigned MaxCachelinesPerPrefetch = 32;
-  constexpr unsigned CacheLineSize = 128;
-  constexpr uint64_t MaxPrefetchSize = 64 * 1024;
+  constexpr uint64_t PrefetchInstSize = 8;
 
-  uint64_t PrefetchSize = std::min(CodeSizeInBytes, MaxPrefetchSize);
-  uint64_t Offset = 0;
-  uint64_t Remaining = PrefetchSize;
+  uint64_t FirstPrefetch = calcFirstPrefetchOffset(PrefetchBlockOffset);
+  uint64_t PrefetchSize =
+      calcPrefetchSize(CodeSizeInBytes, PrefetchBlockOffset);
 
-  for (uint64_t I = 0; I < SlotIndex; ++I) {
-    if (Remaining == 0)
-      break;
-    uint64_t Cachelines =
-        std::min(divideCeil(Remaining, CacheLineSize),
-                 static_cast<uint64_t>(MaxCachelinesPerPrefetch));
-    uint64_t PrefetchBytes = Cachelines * CacheLineSize;
-    Offset += PrefetchBytes;
-    Remaining = (Remaining > PrefetchBytes) ? Remaining - PrefetchBytes : 0;
+  uint64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
+  if (SlotOffset >= PrefetchSize) {
+    // Instruction semantics adds one to sdata when calculating the length of
+    // the prefetch. This means that even a prefetch instruction with sdata == 0
+    // still performs a prefetch. Therefore, to make this prefetch neutral, we
+    // let the prefetch instruction "prefetch" its own cache line.
+    // TODO: Check if we can replace this with proper s_nop instead.
+    Res = MCValue::get(static_cast<int64_t>(0));
+    return true;
   }
-
+  // Prefetch is relative to this prefetch instruction's PC.
+  uint64_t PC = PrefetchBlockOffset + (SlotIndex * PrefetchInstSize);
+  uint64_t Target = FirstPrefetch + SlotOffset;
+  uint64_t Offset = Target - PC;
   Res = MCValue::get(static_cast<int64_t>(Offset));
   return true;
 }
@@ -431,13 +455,17 @@ AMDGPUMCExpr::createInstPrefSize(const MCExpr *CodeSizeBytes, MCContext &Ctx) {
 }
 
 const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchCachelines(
-    const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes, MCContext &Ctx) {
-  return create(AGVK_PrefetchCachelines, {SlotIndex, CodeSizeBytes}, Ctx);
+    const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
+    const MCExpr *PrefetchBlockOffset, MCContext &Ctx) {
+  return create(AGVK_PrefetchCachelines,
+                {SlotIndex, CodeSizeBytes, PrefetchBlockOffset}, Ctx);
 }
 
 const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchOffset(
-    const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes, MCContext &Ctx) {
-  return create(AGVK_PrefetchOffset, {SlotIndex, CodeSizeBytes}, Ctx);
+    const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
+    const MCExpr *PrefetchBlockOffset, MCContext &Ctx) {
+  return create(AGVK_PrefetchOffset,
+                {SlotIndex, CodeSizeBytes, PrefetchBlockOffset}, Ctx);
 }
 
 const AMDGPUMCExpr *AMDGPUMCExpr::createLit(LitModifier Lit, int64_t Value,
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index d3693fb14045e..0db84e172f2ed 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -124,20 +124,24 @@ class AMDGPUMCExpr : public MCTargetExpr {
   /// slot.
   /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
   /// CodeSizeBytes is the total code size in bytes.
+  /// PrefetchBlockOffset is the byte offset of the first prefetch instruction
+  /// from the function entry.
   /// Returns the requested cacheline count minus one, encoded for the 5-bit
   /// sdata field.
   static const AMDGPUMCExpr *
   createPrefetchCachelines(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
-                           MCContext &Ctx);
+                           const MCExpr *PrefetchBlockOffset, MCContext &Ctx);
 
   /// Create an expression for computing the byte offset for a prefetch slot.
   /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
   /// CodeSizeBytes is the total code size in bytes.
-  /// Returns the cumulative byte offset where this slot should start
-  /// prefetching.
-  static const AMDGPUMCExpr *createPrefetchOffset(const MCExpr *SlotIndex,
-                                                  const MCExpr *CodeSizeBytes,
-                                                  MCContext &Ctx);
+  /// PrefetchBlockOffset is the byte offset of the first prefetch instruction
+  /// from the function entry.
+  /// Returns the byte offset from the PC of the corresponding prefetch
+  /// instruction to where this slot should start prefetching.
+  static const AMDGPUMCExpr *
+  createPrefetchOffset(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
+                       const MCExpr *PrefetchBlockOffset, MCContext &Ctx);
 
   static const AMDGPUMCExpr *createLit(LitModifier Lit, int64_t Value,
                                        MCContext &Ctx);

>From 1e3a6075c53f09def8344c2b95ac70cedb39ce14 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Tue, 4 Aug 2026 02:42:18 -0500
Subject: [PATCH 06/22] Update icache prefetch test

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/test/CodeGen/AMDGPU/icache-prefetch.ll | 206 +++++++++++++-------
 1 file changed, 139 insertions(+), 67 deletions(-)

diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 2dc738cd703a6..45c2c6d08d0ce 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -8,105 +8,177 @@
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
 ; RUN: llvm-objdump -d %t.o | FileCheck -check-prefix=GFX1250-OBJ %s
 
-; Test that ICache prefetch instructions are inserted for entry functions on gfx1250.
-; Verify object file has resolved cacheline counts.
-; GFX1250-OBJ-LABEL: <entry_kernel>:
-; GFX1250-OBJ:       s_prefetch_inst_pc_rel 0x0, null, 2
-; GFX1250-OBJ-NEXT:  s_prefetch_inst_pc_rel 0x100, null, 0
+; Use .space to make the estimated MachineFunction size and final assembled
+; size large without spelling out thousands of instructions.
 
-define amdgpu_kernel void @entry_kernel(ptr addrspace(1) %out) {
-; GFX1250-LABEL: entry_kernel:
+; A function below the 32 KiB kernel-descriptor prefetch limit must not use
+; explicit prefetch instructions.
+; GFX1250-OBJ-LABEL: <below_threshold>:
+; GFX1250-OBJ-NOT:   s_prefetch_inst_pc_rel
+define amdgpu_kernel void @below_threshold() {
+; GFX1250-LABEL: below_threshold:
 ; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    ;;#ASMSTART
+; GFX1250-NEXT:    .space 30000
+; GFX1250-NEXT:    ;;#ASMEND
+; GFX1250-NEXT:    s_endpgm
+;
+; NO-PREFETCH-LABEL: below_threshold:
+; NO-PREFETCH:       ; %bb.0:
+; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; NO-PREFETCH-NEXT:    v_nop
+; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT:    ;;#ASMSTART
+; NO-PREFETCH-NEXT:    .space 30000
+; NO-PREFETCH-NEXT:    ;;#ASMEND
+; NO-PREFETCH-NEXT:    s_endpgm
+  call void asm sideeffect ".space 30000", ""()
+  ret void
+}
+
+; Exercise several full 4 KiB prefetch slots and a partial final slot.
+; GFX1250-OBJ-LABEL:      <partial_final_slot>:
+; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x80, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x1078, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2070, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3068, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4060, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5058, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6050, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7048, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8040, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9038, null, 24
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x0, null, 0
+define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
+; GFX1250-LABEL: partial_final_slot:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:  .Lpref_block_start0:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(0, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(1, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(2, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(3, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(4, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(5, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(6, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(7, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(8, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(9, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(10, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 24, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(0, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(1, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(2, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(3, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(4, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(5, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(6, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(7, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(8, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(9, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(10, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(11, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(12, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(13, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(14, .Lpref_func_end0-entry_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end0-entry_kernel), null, prefetchcachelines(15, .Lpref_func_end0-entry_kernel)
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
 ; GFX1250-NEXT:    v_mov_b32_e32 v0, 0
+; GFX1250-NEXT:    ;;#ASMSTART
+; GFX1250-NEXT:    .space 40000
+; GFX1250-NEXT:    ;;#ASMEND
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_store_b32 v0, v0, s[0:1]
 ; GFX1250-NEXT:    s_endpgm
 ; GFX1250-NEXT:  .Lpref_func_end0:
 ;
-; NO-PREFETCH-LABEL: entry_kernel:
+; NO-PREFETCH-LABEL: partial_final_slot:
 ; NO-PREFETCH:       ; %bb.0:
+; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; NO-PREFETCH-NEXT:    v_nop
 ; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
 ; NO-PREFETCH-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
 ; NO-PREFETCH-NEXT:    v_mov_b32_e32 v0, 0
+; NO-PREFETCH-NEXT:    ;;#ASMSTART
+; NO-PREFETCH-NEXT:    .space 40000
+; NO-PREFETCH-NEXT:    ;;#ASMEND
 ; NO-PREFETCH-NEXT:    s_wait_kmcnt 0x0
 ; NO-PREFETCH-NEXT:    global_store_b32 v0, v0, s[0:1]
 ; NO-PREFETCH-NEXT:    s_endpgm
+  call void asm sideeffect ".space 40000", ""()
   store i32 0, ptr addrspace(1) %out
   ret void
 }
 
-; Verify that called functions do NOT get prefetch instructions.
+; Exercise the 16-instruction limit and reserve the cache line containing the
+; first prefetch instruction instead of attempting to replace all 64 KiB.
+; GFX1250-OBJ-LABEL:      <cache_size_limit>:
+; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x80, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x1078, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2070, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3068, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4060, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5058, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6050, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7048, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8040, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9038, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xa030, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xb028, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xc020, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xd018, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xe010, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xf008, null, 30
+define amdgpu_kernel void @cache_size_limit() {
+; GFX1250-LABEL: cache_size_limit:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:  .Lpref_block_start1:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(0, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(1, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(2, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(3, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(4, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(5, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(6, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(7, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(8, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(9, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(10, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(11, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(12, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(13, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(14, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(15, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    ;;#ASMSTART
+; GFX1250-NEXT:    .space 65536
+; GFX1250-NEXT:    ;;#ASMEND
+; GFX1250-NEXT:    s_endpgm
+; GFX1250-NEXT:  .Lpref_func_end1:
+;
+; NO-PREFETCH-LABEL: cache_size_limit:
+; NO-PREFETCH:       ; %bb.0:
+; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; NO-PREFETCH-NEXT:    v_nop
+; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT:    ;;#ASMSTART
+; NO-PREFETCH-NEXT:    .space 65536
+; NO-PREFETCH-NEXT:    ;;#ASMEND
+; NO-PREFETCH-NEXT:    s_endpgm
+  call void asm sideeffect ".space 65536", ""()
+  ret void
+}
 
-define void @called_function(ptr addrspace(1) %out) {
+; Non-entry functions must not get explicit prefetch instructions regardless
+; of their size.
+define void @called_function() {
 ; GFX1250-LABEL: called_function:
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_mov_b32_e32 v2, 1
-; GFX1250-NEXT:    global_store_b32 v[0:1], v2, off
+; GFX1250-NEXT:    ;;#ASMSTART
+; GFX1250-NEXT:    .space 40000
+; GFX1250-NEXT:    ;;#ASMEND
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
 ;
 ; NO-PREFETCH-LABEL: called_function:
 ; NO-PREFETCH:       ; %bb.0:
 ; NO-PREFETCH-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; NO-PREFETCH-NEXT:    s_wait_kmcnt 0x0
-; NO-PREFETCH-NEXT:    v_mov_b32_e32 v2, 1
-; NO-PREFETCH-NEXT:    global_store_b32 v[0:1], v2, off
+; NO-PREFETCH-NEXT:    ;;#ASMSTART
+; NO-PREFETCH-NEXT:    .space 40000
+; NO-PREFETCH-NEXT:    ;;#ASMEND
 ; NO-PREFETCH-NEXT:    s_set_pc_i64 s[30:31]
-  store i32 1, ptr addrspace(1) %out
-  ret void
-}
-
-; GFX1250-OBJ-LABEL: <tiny_kernel>:
-; GFX1250-OBJ:       s_prefetch_inst_pc_rel 0x0, null, 2
-; GFX1250-OBJ-NEXT:  s_prefetch_inst_pc_rel 0x100, null, 0
-
-define amdgpu_kernel void @tiny_kernel() {
-; GFX1250-LABEL: tiny_kernel:
-; GFX1250:       ; %bb.0:
-; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 24, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(0, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(1, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(2, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(3, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(4, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(5, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(6, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(7, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(8, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(9, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(10, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(11, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(12, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(13, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(14, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-tiny_kernel), null, prefetchcachelines(15, .Lpref_func_end1-tiny_kernel)
-; GFX1250-NEXT:    s_endpgm
-; GFX1250-NEXT:  .Lpref_func_end1:
-;
-; NO-PREFETCH-LABEL: tiny_kernel:
-; NO-PREFETCH:       ; %bb.0:
-; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; NO-PREFETCH-NEXT:    s_endpgm
+  call void asm sideeffect ".space 40000", ""()
   ret void
 }

>From 02f7aa5ffa52b47c85741d04c87cb11fb8f369a7 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Tue, 4 Aug 2026 03:39:21 -0500
Subject: [PATCH 07/22] Extract instruction prefetch into separate pass

Passes running after the pre-emit peephole pass can have a significant
impact on code size. Extract the instruction cache prefetching into a
separate pass, so it can run after those passes and insert based on a
more reliable code size estimation.

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/AMDGPU.h               |  10 ++
 .../AMDGPU/AMDGPUInsertICachePrefetch.cpp     | 145 ++++++++++++++++++
 llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def |   1 +
 .../lib/Target/AMDGPU/AMDGPUTargetMachine.cpp |   7 +
 llvm/lib/Target/AMDGPU/CMakeLists.txt         |   1 +
 llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp  |  97 ------------
 llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll  |   2 +
 llvm/test/CodeGen/AMDGPU/llc-pipeline.ll      |   4 +
 8 files changed, 170 insertions(+), 97 deletions(-)
 create mode 100644 llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp

diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.h b/llvm/lib/Target/AMDGPU/AMDGPU.h
index 809540bd05b45..c117fe4083809 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.h
@@ -370,6 +370,13 @@ struct AMDGPUInsertDelayAluPass
                         MachineFunctionAnalysisManager &MFAM);
 };
 
+class AMDGPUInsertICachePrefetchPass
+    : public RequiredPassInfoMixin<AMDGPUInsertICachePrefetchPass> {
+public:
+  PreservedAnalyses run(MachineFunction &MF,
+                        MachineFunctionAnalysisManager &MFAM);
+};
+
 FunctionPass *createAMDGPUISelDag(TargetMachine &TM, CodeGenOptLevel OptLevel);
 ModulePass *createAMDGPUAlwaysInlinePass(bool GlobalOpt = true);
 
@@ -611,6 +618,9 @@ extern char &SIModeRegisterID;
 void initializeAMDGPUInsertDelayAluLegacyPass(PassRegistry &);
 extern char &AMDGPUInsertDelayAluID;
 
+void initializeAMDGPUInsertICachePrefetchLegacyPass(PassRegistry &);
+extern char &AMDGPUInsertICachePrefetchID;
+
 void initializeAMDGPULowerVGPREncodingLegacyPass(PassRegistry &);
 extern char &AMDGPULowerVGPREncodingLegacyID;
 
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
new file mode 100644
index 0000000000000..119db87ac0a46
--- /dev/null
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -0,0 +1,145 @@
+//===- AMDGPUInsertICachePrefetch.cpp - Insert ICache prefetches ---------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+/// \file
+/// Insert instruction-cache prefetches for large AMDHSA entry functions.
+//
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPU.h"
+#include "GCNSubtarget.h"
+#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
+#include "SIMachineFunctionInfo.h"
+#include "SIProgramInfo.h"
+#include "llvm/CodeGen/MachineFunctionPass.h"
+#include "llvm/CodeGen/MachineInstrBuilder.h"
+#include "llvm/Support/CommandLine.h"
+#include "llvm/TargetParser/Triple.h"
+
+using namespace llvm;
+
+#define DEBUG_TYPE "amdgpu-insert-icache-prefetch"
+
+static cl::opt<bool>
+    EnableICachePrefetch("amdgpu-icache-prefetch",
+                         cl::desc("Insert ICache prefetch instructions"),
+                         cl::init(true), cl::Hidden);
+
+namespace {
+
+class AMDGPUInsertICachePrefetch {
+public:
+  bool run(MachineFunction &MF);
+};
+
+class AMDGPUInsertICachePrefetchLegacy : public MachineFunctionPass {
+public:
+  static char ID;
+
+  AMDGPUInsertICachePrefetchLegacy() : MachineFunctionPass(ID) {}
+
+  void getAnalysisUsage(AnalysisUsage &AU) const override {
+    AU.setPreservesCFG();
+    MachineFunctionPass::getAnalysisUsage(AU);
+  }
+
+  bool runOnMachineFunction(MachineFunction &MF) override {
+    return AMDGPUInsertICachePrefetch().run(MF);
+  }
+};
+
+} // end anonymous namespace
+
+bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
+  if (!EnableICachePrefetch)
+    return false;
+
+  const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
+  if (!ST.hasICachePrefetch())
+    return false;
+
+  // Only run for AMDHSA - this is where kernel descriptors are used and
+  // rsrc3 INST_PREF_SIZE is relevant.
+  if (ST.getTargetTriple().getOS() != Triple::AMDHSA)
+    return false;
+
+  SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
+  if (!MFI->isEntryFunction())
+    return false;
+
+  SIProgramInfo PI;
+  uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
+  // The kernel descriptor can specify an instruction prefetch size of up to 256
+  // in INST_PREF_SIZE. At a granularity of 128B, this equals 32KiB of
+  // instructions that can be prefetched without inserting explicit prefetch
+  // instructions.
+  constexpr uint64_t MaxKDPrefetch = 1u << 15;
+  if (ProgramSize <= MaxKDPrefetch)
+    return false;
+
+  MachineBasicBlock &EntryBB = MF.front();
+  MachineBasicBlock::iterator InsertPt = EntryBB.begin();
+
+  // Skip past any instructions that must remain at the very beginning:
+  // - Debug values and CFI instructions
+  // - S_SETREG_IMM32_B32 instructions that set up MODE register bits
+  //   (e.g., REPLAY_MODE bit 25 from SIFrameLowering)
+  // We want the prefetches to come after all initial MODE setup.
+  while (InsertPt != EntryBB.end()) {
+    if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction() ||
+        InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32) {
+      ++InsertPt;
+      continue;
+    }
+    break;
+  }
+
+  const SIInstrInfo *TII = ST.getInstrInfo();
+  DebugLoc DL;
+
+  // Each prefetch can transfer 4KiB of instructions. Retain the existing
+  // slack for growth after this late pass, such as padding and alignment. The
+  // offset and sdata operands are placeholders; AMDGPUAsmPrinter fixes them
+  // up using the exact emitted code size.
+  constexpr uint64_t PrefetchSlack = 2 * 1024;
+  constexpr uint64_t BytesPerPrefetch = 4 * 1024;
+  // Each prefetch can transfer up to 32 cachelines of 128 bytes = 4KiB.
+  // 16 instructions cover 64KiB (the full ICache size).
+  constexpr unsigned MaxNumPrefetchInsts = 16;
+  unsigned NumPrefetches =
+      llvm::divideCeil(ProgramSize + PrefetchSlack, BytesPerPrefetch);
+  NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
+  for (unsigned I = 0; I < NumPrefetches; ++I) {
+    BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
+        .addImm(0)                 // offset (placeholder, fixed up later)
+        .addReg(AMDGPU::SGPR_NULL) // soffset
+        .addImm(I);                // sdata (slot index, fixed up later)
+  }
+
+  // The instruction is fire-and-forget: it updates no wave wait counter and
+  // returns neither a result nor an error. It is therefore safe to insert
+  // after the wait-counter and hazard passes.
+  MFI->setHasICachePrefetch(true);
+  return true;
+}
+
+PreservedAnalyses
+llvm::AMDGPUInsertICachePrefetchPass::run(MachineFunction &MF,
+                                          MachineFunctionAnalysisManager &) {
+  if (!AMDGPUInsertICachePrefetch().run(MF))
+    return PreservedAnalyses::all();
+  auto PA = getMachineFunctionPassPreservedAnalyses();
+  PA.preserveSet<CFGAnalyses>();
+  return PA;
+}
+
+char AMDGPUInsertICachePrefetchLegacy::ID = 0;
+char &llvm::AMDGPUInsertICachePrefetchID = AMDGPUInsertICachePrefetchLegacy::ID;
+
+INITIALIZE_PASS(AMDGPUInsertICachePrefetchLegacy, DEBUG_TYPE,
+                "AMDGPU Insert ICache Prefetch", false, false)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def b/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def
index 372d5f5acab21..883dd4eab455b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPassRegistry.def
@@ -119,6 +119,7 @@ MACHINE_FUNCTION_PASS("amdgpu-asm-printer", AMDGPUAsmPrinterPass())
 MACHINE_FUNCTION_PASS("amdgpu-global-isel-divergence-lowering",
                       AMDGPUGlobalISelDivergenceLoweringPass())
 MACHINE_FUNCTION_PASS("amdgpu-insert-delay-alu", AMDGPUInsertDelayAluPass())
+MACHINE_FUNCTION_PASS("amdgpu-insert-icache-prefetch", AMDGPUInsertICachePrefetchPass())
 MACHINE_FUNCTION_PASS("amdgpu-isel", AMDGPUISelDAGToDAGPass(*this))
 MACHINE_FUNCTION_PASS("amdgpu-lower-vgpr-encoding", AMDGPULowerVGPREncodingPass())
 MACHINE_FUNCTION_PASS("amdgpu-mark-last-scratch-load", AMDGPUMarkLastScratchLoadPass())
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
index 72c028edbaef5..05a5f4545fcae 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
@@ -739,6 +739,7 @@ extern "C" LLVM_ABI LLVM_EXTERNAL_VISIBILITY void LLVMInitializeAMDGPUTarget() {
   initializeAMDGPURewriteUndefForPHILegacyPass(*PR);
   initializeSIAnnotateControlFlowLegacyPass(*PR);
   initializeAMDGPUInsertDelayAluLegacyPass(*PR);
+  initializeAMDGPUInsertICachePrefetchLegacyPass(*PR);
   initializeAMDGPULowerVGPREncodingLegacyPass(*PR);
   initializeSIInsertHardClausesLegacyPass(*PR);
   initializeSIInsertWaitcntsLegacyPass(*PR);
@@ -2066,6 +2067,9 @@ void GCNPassConfig::addPreEmitPass() {
   if (isPassEnabled(EnableInsertDelayAlu, CodeGenOptLevel::Less))
     addPass(&AMDGPUInsertDelayAluID);
 
+  if (getOptLevel() > CodeGenOptLevel::None)
+    addPass(&AMDGPUInsertICachePrefetchID);
+
   addPass(&BranchRelaxationPassID);
 }
 
@@ -2815,6 +2819,9 @@ void AMDGPUCodeGenPassBuilder::addPreEmitPass(PassManagerWrapper &PMW) {
     addMachineFunctionPass(AMDGPUInsertDelayAluPass(), PMW);
   }
 
+  if (TM.getOptLevel() > CodeGenOptLevel::None)
+    addMachineFunctionPass(AMDGPUInsertICachePrefetchPass(), PMW);
+
   addMachineFunctionPass(BranchRelaxationPass(), PMW);
 }
 
diff --git a/llvm/lib/Target/AMDGPU/CMakeLists.txt b/llvm/lib/Target/AMDGPU/CMakeLists.txt
index 4a5f77d55afa5..1bbdbc9ae2786 100644
--- a/llvm/lib/Target/AMDGPU/CMakeLists.txt
+++ b/llvm/lib/Target/AMDGPU/CMakeLists.txt
@@ -63,6 +63,7 @@ add_llvm_target(AMDGPUCodeGen
   AMDGPUHSAMetadataStreamer.cpp
   AMDGPUHWEvents.cpp
   AMDGPUInsertDelayAlu.cpp
+  AMDGPUInsertICachePrefetch.cpp
   AMDGPUInstCombineIntrinsic.cpp
   AMDGPUUniformIntrinsicCombine.cpp
   AMDGPUInstrInfo.cpp
diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index 11344ff8e4709..e5b1de7bc4ec9 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -20,9 +20,6 @@
 
 #include "AMDGPU.h"
 #include "GCNSubtarget.h"
-#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
-#include "SIMachineFunctionInfo.h"
-#include "SIProgramInfo.h"
 #include "llvm/ADT/Statistic.h"
 #include "llvm/CodeGen/MachineDominators.h"
 #include "llvm/CodeGen/MachineFunctionPass.h"
@@ -30,8 +27,6 @@
 #include "llvm/CodeGen/MachineLoopInfo.h"
 #include "llvm/CodeGen/TargetSchedule.h"
 #include "llvm/Support/BranchProbability.h"
-#include "llvm/Support/CommandLine.h"
-#include "llvm/TargetParser/Triple.h"
 using namespace llvm;
 
 #define DEBUG_TYPE "si-pre-emit-peephole"
@@ -51,13 +46,6 @@ struct ModeFieldState {
   bool isTracked() const { return PendingWrite || Value; }
 };
 
-static cl::opt<bool>
-    EnableICachePrefetch("amdgpu-icache-prefetch",
-                         cl::desc("Insert ICache prefetch instructions"),
-                         cl::init(true), cl::Hidden);
-
-namespace {
-
 class SIPreEmitPeephole {
 private:
   const GCNSubtarget *ST = nullptr;
@@ -65,7 +53,6 @@ class SIPreEmitPeephole {
   const SIRegisterInfo *TRI = nullptr;
   MachineLoopInfo *MLI = nullptr;
 
-  bool insertICachePrefetch(MachineFunction &MF);
   bool optimizeVccBranch(MachineInstr &MI) const;
   void updateMLIBeforeRemovingEdge(MachineBasicBlock *From,
                                    MachineBasicBlock *To) const;
@@ -864,86 +851,6 @@ MachineInstrBuilder SIPreEmitPeephole::createUnpackedMI(MachineInstr &I,
   return NewMI;
 }
 
-bool SIPreEmitPeephole::insertICachePrefetch(MachineFunction &MF) {
-  // Only run on targets that support ICache prefetching.
-  if (!ST->hasICachePrefetch())
-    return false;
-
-  // Only run for AMDHSA - this is where kernel descriptors are used
-  // and rsrc3 INST_PREF_SIZE is relevant.
-  const Triple &TT = ST->getTargetTriple();
-  if (TT.getOS() != Triple::AMDHSA)
-    return false;
-
-  SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
-
-  // Only insert prefetch instructions for entry functions.
-  if (!MFI->isEntryFunction())
-    return false;
-
-  SIProgramInfo PI;
-  uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
-  // The kernel descriptor can specify an instruction prefetch size of up to 256
-  // in INST_PREF_SIZE. At a granularity of 128B, this equals 32KiB of
-  // instructions that can be prefetched without inserting explicit prefetch
-  // instructions.
-  constexpr uint64_t MaxKDPrefetch = 1u << 15;
-  if (ProgramSize <= MaxKDPrefetch)
-    return false;
-
-  MachineBasicBlock &EntryBB = MF.front();
-  MachineBasicBlock::iterator InsertPt = EntryBB.begin();
-
-  // Skip past any instructions that must remain at the very beginning:
-  // - Debug values and CFI instructions
-  // - S_SETREG_IMM32_B32 instructions that set up MODE register bits
-  //   (e.g., REPLAY_MODE bit 25 from SIFrameLowering)
-  // We want the prefetches to come after all initial MODE setup.
-  while (InsertPt != EntryBB.end()) {
-    if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction()) {
-      ++InsertPt;
-      continue;
-    }
-    if (InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32) {
-      ++InsertPt;
-      continue;
-    }
-    break;
-  }
-
-  DebugLoc DL;
-
-  // Calculate the number of prefetch instructions required for the current
-  // program size. Each prefetch can transfer 4KiB of instructions. Add
-  // some slack as inserting the prefetches and later transformations, e.g.,
-  // padding and alignment, will introduce additional bytes. The offset and
-  // sdata operands are placeholders - the sdata operand stores the slot index
-  // (0-15). Both will be fixed up in AMDGPUAsmPrinter based on the actual code
-  // size.
-  constexpr uint64_t PrefetchSlack = 2 * 1024;
-  constexpr uint64_t BytesPerPrefetch = 4 * 1024;
-  // Each prefetch can transfer up to 32 cachelines of 128 bytes = 4KiB.
-  // 16 instructions cover 64KiB (the full ICache size).
-  constexpr unsigned MaxNumPrefetchInsts = 16;
-  unsigned NumPrefetches =
-      llvm::divideCeil(ProgramSize + PrefetchSlack, BytesPerPrefetch);
-  // Limit to 16 prefetches at most, otherwise we'd exceed the cache size.
-  NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
-  for (unsigned I = 0; I < NumPrefetches; ++I) {
-    BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
-        .addImm(0)                 // offset (placeholder, fixed up later)
-        .addReg(AMDGPU::SGPR_NULL) // soffset
-        .addImm(I);                // sdata (slot index, fixed up later)
-  }
-
-  // Mark that we've inserted ICache prefetch instructions.
-  // This tells AsmPrinter to set rsrc3 INST_PREF_SIZE to 1 and fix up
-  // the prefetch cacheline counts.
-  MFI->setHasICachePrefetch(true);
-
-  return true;
-}
-
 PreservedAnalyses
 llvm::SIPreEmitPeepholePass::run(MachineFunction &MF,
                                  MachineFunctionAnalysisManager &MFAM) {
@@ -966,10 +873,6 @@ bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
   MLI = LoopInfo;
   bool Changed = false;
 
-  // Insert ICache prefetch instructions if enabled.
-  if (EnableICachePrefetch)
-    Changed |= insertICachePrefetch(MF);
-
   MF.RenumberBlocks();
 
   for (MachineBasicBlock &MBB : MF) {
diff --git a/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll b/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll
index cacab31f263c8..56b61ac7ea90d 100644
--- a/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll
+++ b/llvm/test/CodeGen/AMDGPU/llc-pipeline-npm.ll
@@ -284,6 +284,7 @@
 ; GCN-O2-NEXT:       amdgpu-wait-sgpr-hazards
 ; GCN-O2-NEXT:       amdgpu-lower-vgpr-encoding
 ; GCN-O2-NEXT:       amdgpu-insert-delay-alu
+; GCN-O2-NEXT:       amdgpu-insert-icache-prefetch
 ; GCN-O2-NEXT:       branch-relaxation
 ; GCN-O2-NEXT:       reg-usage-collector
 ; GCN-O2-NEXT:       remove-loads-into-fake-uses
@@ -473,6 +474,7 @@
 ; GCN-O3-NEXT:       amdgpu-wait-sgpr-hazards
 ; GCN-O3-NEXT:       amdgpu-lower-vgpr-encoding
 ; GCN-O3-NEXT:       amdgpu-insert-delay-alu
+; GCN-O3-NEXT:       amdgpu-insert-icache-prefetch
 ; GCN-O3-NEXT:       branch-relaxation
 ; GCN-O3-NEXT:       reg-usage-collector
 ; GCN-O3-NEXT:       remove-loads-into-fake-uses
diff --git a/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll b/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
index 59c74e1b18c9c..8f88d84280b47 100644
--- a/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
+++ b/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
@@ -455,6 +455,7 @@
 ; GCN-O1-NEXT:        AMDGPU Insert waits for SGPR read hazards
 ; GCN-O1-NEXT:        AMDGPU Lower VGPR Encoding
 ; GCN-O1-NEXT:        AMDGPU Insert Delay ALU
+; GCN-O1-NEXT:        AMDGPU Insert ICache Prefetch
 ; GCN-O1-NEXT:        Branch relaxation pass
 ; GCN-O1-NEXT:        Register Usage Information Collector Pass
 ; GCN-O1-NEXT:        Remove Loads Into Fake Uses
@@ -785,6 +786,7 @@
 ; GCN-O1-OPTS-NEXT:        AMDGPU Insert waits for SGPR read hazards
 ; GCN-O1-OPTS-NEXT:        AMDGPU Lower VGPR Encoding
 ; GCN-O1-OPTS-NEXT:        AMDGPU Insert Delay ALU
+; GCN-O1-OPTS-NEXT:        AMDGPU Insert ICache Prefetch
 ; GCN-O1-OPTS-NEXT:        Branch relaxation pass
 ; GCN-O1-OPTS-NEXT:        Register Usage Information Collector Pass
 ; GCN-O1-OPTS-NEXT:        Remove Loads Into Fake Uses
@@ -1120,6 +1122,7 @@
 ; GCN-O2-NEXT:        AMDGPU Insert waits for SGPR read hazards
 ; GCN-O2-NEXT:        AMDGPU Lower VGPR Encoding
 ; GCN-O2-NEXT:        AMDGPU Insert Delay ALU
+; GCN-O2-NEXT:        AMDGPU Insert ICache Prefetch
 ; GCN-O2-NEXT:        Branch relaxation pass
 ; GCN-O2-NEXT:        Register Usage Information Collector Pass
 ; GCN-O2-NEXT:        Remove Loads Into Fake Uses
@@ -1470,6 +1473,7 @@
 ; GCN-O3-NEXT:        AMDGPU Insert waits for SGPR read hazards
 ; GCN-O3-NEXT:        AMDGPU Lower VGPR Encoding
 ; GCN-O3-NEXT:        AMDGPU Insert Delay ALU
+; GCN-O3-NEXT:        AMDGPU Insert ICache Prefetch
 ; GCN-O3-NEXT:        Branch relaxation pass
 ; GCN-O3-NEXT:        Register Usage Information Collector Pass
 ; GCN-O3-NEXT:        Remove Loads Into Fake Uses

>From 3dc1282ecc4c4c87421fa24287f27d2c18345890 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Wed, 5 Aug 2026 02:21:36 -0500
Subject: [PATCH 08/22] Skip unclaused VMEM prologue for insertion

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 .../AMDGPU/AMDGPUInsertICachePrefetch.cpp     | 14 ++++
 llvm/test/CodeGen/AMDGPU/icache-prefetch.ll   | 64 +++++++++----------
 2 files changed, 46 insertions(+), 32 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index 119db87ac0a46..213fce4295d7f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -87,15 +87,29 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
 
   // Skip past any instructions that must remain at the very beginning:
   // - Debug values and CFI instructions
+  // - The gfx1250 initial unclaused-VMEM workaround
   // - S_SETREG_IMM32_B32 instructions that set up MODE register bits
   //   (e.g., REPLAY_MODE bit 25 from SIFrameLowering)
   // We want the prefetches to come after all initial MODE setup.
+  bool SkippedInitialUnclausedVmemPrologue =
+      !ST.hasRequiresInitialUnclausedVmem();
   while (InsertPt != EntryBB.end()) {
     if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction() ||
         InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32) {
       ++InsertPt;
       continue;
     }
+    if (!SkippedInitialUnclausedVmemPrologue &&
+        InsertPt->getOpcode() == AMDGPU::GLOBAL_PREFETCH_B8_SADDR) {
+      auto Next = InsertPt;
+      ++Next;
+      if (Next != EntryBB.end() &&
+          Next->getOpcode() == AMDGPU::V_NOP_e32) {
+        InsertPt = ++Next;
+        SkippedInitialUnclausedVmemPrologue = true;
+        continue;
+      }
+    }
     break;
   }
 
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 45c2c6d08d0ce..18777c160cc4e 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -41,20 +41,23 @@ define amdgpu_kernel void @below_threshold() {
 
 ; Exercise several full 4 KiB prefetch slots and a partial final slot.
 ; GFX1250-OBJ-LABEL:      <partial_final_slot>:
-; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x80, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x1078, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2070, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3068, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4060, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5058, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6050, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7048, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8040, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9038, null, 24
+; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x68, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x1060, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2058, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3050, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4048, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5040, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6038, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7030, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8028, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9020, null, 24
 ; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x0, null, 0
 define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; GFX1250-LABEL: partial_final_slot:
 ; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
 ; GFX1250-NEXT:  .Lpref_block_start0:
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(0, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(1, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
@@ -67,9 +70,6 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(8, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(9, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(10, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; GFX1250-NEXT:    v_nop
-; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
 ; GFX1250-NEXT:    v_mov_b32_e32 v0, 0
 ; GFX1250-NEXT:    ;;#ASMSTART
@@ -101,25 +101,28 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; Exercise the 16-instruction limit and reserve the cache line containing the
 ; first prefetch instruction instead of attempting to replace all 64 KiB.
 ; GFX1250-OBJ-LABEL:      <cache_size_limit>:
-; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x80, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x1078, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2070, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3068, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4060, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5058, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6050, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7048, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8040, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9038, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xa030, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xb028, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xc020, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xd018, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xe010, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xf008, null, 30
+; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x68, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x1060, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2058, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3050, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4048, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5040, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6038, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7030, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8028, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9020, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xa018, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xb010, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xc008, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xd000, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xdff8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xeff0, null, 30
 define amdgpu_kernel void @cache_size_limit() {
 ; GFX1250-LABEL: cache_size_limit:
 ; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
 ; GFX1250-NEXT:  .Lpref_block_start1:
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(0, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(1, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
@@ -137,9 +140,6 @@ define amdgpu_kernel void @cache_size_limit() {
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(13, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(14, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(15, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; GFX1250-NEXT:    v_nop
-; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
 ; GFX1250-NEXT:    ;;#ASMSTART
 ; GFX1250-NEXT:    .space 65536
 ; GFX1250-NEXT:    ;;#ASMEND

>From 0602d751240c3c66fbfa027aa1f7bff70ced701e Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 6 Aug 2026 13:55:38 -0500
Subject: [PATCH 09/22] Distribute prefetch instructions across CFG

Instead of inserting all prefetch instructions in the entry block,
distribute them across the entry block and its post-domination chain.
Each block will only prefetch as much code as is needed before the next
candidate, plus some slack to account for latency.

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp   |  10 +-
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h     |   6 -
 .../AMDGPU/AMDGPUInsertICachePrefetch.cpp     | 170 +++++++++---
 llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp  |  18 +-
 .../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp      |  65 ++---
 .../Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h |  17 +-
 llvm/lib/Target/AMDGPU/SIProgramInfo.cpp      |  23 +-
 llvm/lib/Target/AMDGPU/SIProgramInfo.h        |   6 +
 llvm/test/CodeGen/AMDGPU/icache-prefetch.ll   | 257 ++++++++++++++----
 9 files changed, 395 insertions(+), 177 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 25cd2516e145c..fcbc576364e3f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -193,13 +193,10 @@ void AMDGPUAsmPrinter::emitFunctionBodyStart() {
   const GCNSubtarget &STM = MF->getSubtarget<GCNSubtarget>();
   const Function &F = MF->getFunction();
 
-  // If ICache prefetch is enabled, create the function end symbol and prefetch
-  // block start symbol early so it can be referenced by the prefetch MCExprs
-  // during instruction emission.
-  if (MFI.hasICachePrefetch()) {
+  // If ICache prefetch is enabled, create the function end symbol early so it
+  // can be referenced by the prefetch MCExprs during instruction emission.
+  if (MFI.hasICachePrefetch())
     PrefetchEndSym = createTempSymbol("pref_func_end");
-    PrefetchBlockStartSym = createTempSymbol("pref_block_start");
-  }
 
   // TODO: We're checking this late, would be nice to check it earlier.
   if (STM.requiresCodeObjectV6() && CodeObjectVersion < AMDGPU::AMDHSA_COV6) {
@@ -235,7 +232,6 @@ void AMDGPUAsmPrinter::emitFunctionBodyEnd() {
     OutStreamer->emitLabel(PrefetchEndSym);
   }
   PrefetchEndSym = nullptr;
-  PrefetchBlockStartSym = nullptr;
 }
 
 /// Set bits in a kernel descriptor MCExpr field:
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
index 22eb7da72812b..871bd738b947b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
@@ -60,9 +60,6 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
   // Symbol for the function end, used by ICache prefetch MCExprs.
   // Created early in emitFunctionBodyStart when prefetch is enabled.
   MCSymbol *PrefetchEndSym = nullptr;
-  // Symbol for the first prefetch instruction, used by ICache prefetch MCExprs.
-  // Created early in emitFunctionBodyStart when prefetch is enabled.
-  MCSymbol *PrefetchBlockStartSym = nullptr;
 
   // When appropriate, add a _dvgpr$ symbol.
   void emitDVgprSymbol(MachineFunction &MF);
@@ -167,9 +164,6 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
 
   /// Get the symbol for the function end, used for ICache prefetch MCExprs.
   MCSymbol *getPrefetchEndSym() const { return PrefetchEndSym; }
-  /// Get the symbol for the first prefetch instruction, used for ICache
-  /// prefetch MCExprs.
-  MCSymbol *getPrefetchBlockStartSym() const { return PrefetchBlockStartSym; }
 
   /// Get the code size estimate from SIProgramInfo.
   uint64_t getCodeSize() { return CurrentProgramInfo.getFunctionCodeSize(*MF); }
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index 213fce4295d7f..be8703c4c9d6c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -9,6 +9,10 @@
 /// \file
 /// Insert instruction-cache prefetches for large AMDHSA entry functions.
 //
+// The prefetch instruction is fire-and-forget: it updates no wave wait counter
+// and returns neither a result nor an error. It is therefore safe to insert
+// after the wait-counter and hazard passes.
+//
 //===----------------------------------------------------------------------===//
 
 #include "AMDGPU.h"
@@ -16,8 +20,12 @@
 #include "MCTargetDesc/AMDGPUMCTargetDesc.h"
 #include "SIMachineFunctionInfo.h"
 #include "SIProgramInfo.h"
+#include "llvm/ADT/DenseMap.h"
 #include "llvm/CodeGen/MachineFunctionPass.h"
 #include "llvm/CodeGen/MachineInstrBuilder.h"
+#include "llvm/CodeGen/MachineLoopInfo.h"
+#include "llvm/CodeGen/MachinePostDominators.h"
+#include "llvm/InitializePasses.h"
 #include "llvm/Support/CommandLine.h"
 #include "llvm/TargetParser/Triple.h"
 
@@ -33,7 +41,22 @@ static cl::opt<bool>
 namespace {
 
 class AMDGPUInsertICachePrefetch {
+  MachineLoopInfo &MLI;
+  MachinePostDominatorTree &PDT;
+
+  bool isLoopFreeEntryPostDominator(const MachineBasicBlock &MBB,
+                                    const MachineBasicBlock &EntryBB) const {
+    return !MLI.getLoopFor(&MBB) && PDT.dominates(&MBB, &EntryBB);
+  }
+
 public:
+  // These analyses describe the final machine CFG at this late insertion
+  // point. CFG-based placement will use them to select loop-free blocks on
+  // the entry block's post-dominator chain.
+  AMDGPUInsertICachePrefetch(MachineLoopInfo &MLI,
+                             MachinePostDominatorTree &PDT)
+      : MLI(MLI), PDT(PDT) {}
+
   bool run(MachineFunction &MF);
 };
 
@@ -45,16 +68,57 @@ class AMDGPUInsertICachePrefetchLegacy : public MachineFunctionPass {
 
   void getAnalysisUsage(AnalysisUsage &AU) const override {
     AU.setPreservesCFG();
+    AU.addRequired<MachineLoopInfoWrapperPass>();
+    AU.addRequired<MachinePostDominatorTreeWrapperPass>();
     MachineFunctionPass::getAnalysisUsage(AU);
   }
 
   bool runOnMachineFunction(MachineFunction &MF) override {
-    return AMDGPUInsertICachePrefetch().run(MF);
+    auto &MLI = getAnalysis<MachineLoopInfoWrapperPass>().getLI();
+    auto &PDT =
+        getAnalysis<MachinePostDominatorTreeWrapperPass>().getPostDomTree();
+    return AMDGPUInsertICachePrefetch(MLI, PDT).run(MF);
   }
 };
 
 } // end anonymous namespace
 
+static MachineBasicBlock::iterator
+findMBBInsertionPoint(MachineBasicBlock &MBB, const GCNSubtarget &ST,
+                      bool IsEntryBlock) {
+  MachineBasicBlock::iterator InsertPt = MBB.begin();
+
+  // Skip past any instructions that must remain at the very beginning:
+  // - Debug values and CFI instructions
+  // - In the entry block only, the gfx1250 initial unclaused-VMEM workaround
+  //   and S_SETREG_IMM32_B32 instructions that set up MODE register bits
+  //   (e.g., REPLAY_MODE bit 25 from SIFrameLowering)
+  // In the entry block, we want the prefetches to come after all initial MODE
+  // setup.
+  bool SkippedInitialUnclausedVmemPrologue =
+      !IsEntryBlock || !ST.hasRequiresInitialUnclausedVmem();
+  while (InsertPt != MBB.end()) {
+    if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction() ||
+        (IsEntryBlock &&
+         InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32)) {
+      ++InsertPt;
+      continue;
+    }
+    if (!SkippedInitialUnclausedVmemPrologue &&
+        InsertPt->getOpcode() == AMDGPU::GLOBAL_PREFETCH_B8_SADDR) {
+      auto Next = InsertPt;
+      ++Next;
+      if (Next != MBB.end() && Next->getOpcode() == AMDGPU::V_NOP_e32) {
+        InsertPt = ++Next;
+        SkippedInitialUnclausedVmemPrologue = true;
+        continue;
+      }
+    }
+    break;
+  }
+  return InsertPt;
+}
+
 bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
   if (!EnableICachePrefetch)
     return false;
@@ -82,38 +146,34 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
   if (ProgramSize <= MaxKDPrefetch)
     return false;
 
+  const SIInstrInfo *TII = ST.getInstrInfo();
   MachineBasicBlock &EntryBB = MF.front();
-  MachineBasicBlock::iterator InsertPt = EntryBB.begin();
 
-  // Skip past any instructions that must remain at the very beginning:
-  // - Debug values and CFI instructions
-  // - The gfx1250 initial unclaused-VMEM workaround
-  // - S_SETREG_IMM32_B32 instructions that set up MODE register bits
-  //   (e.g., REPLAY_MODE bit 25 from SIFrameLowering)
-  // We want the prefetches to come after all initial MODE setup.
-  bool SkippedInitialUnclausedVmemPrologue =
-      !ST.hasRequiresInitialUnclausedVmem();
-  while (InsertPt != EntryBB.end()) {
-    if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction() ||
-        InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32) {
-      ++InsertPt;
+  // Walk the post-dominator chain to get candidates in execution order. This
+  // is distinct from the layout order used below to calculate code offsets.
+  SmallVector<MachineBasicBlock *> Candidates = {&EntryBB};
+  for (auto *Node = PDT.getNode(&EntryBB); Node; Node = Node->getIDom()) {
+    MachineBasicBlock *CandBB = Node->getBlock();
+    if (!CandBB)
+      break;
+    if (CandBB == &EntryBB ||
+        !isLoopFreeEntryPostDominator(*CandBB, EntryBB))
       continue;
-    }
-    if (!SkippedInitialUnclausedVmemPrologue &&
-        InsertPt->getOpcode() == AMDGPU::GLOBAL_PREFETCH_B8_SADDR) {
-      auto Next = InsertPt;
-      ++Next;
-      if (Next != EntryBB.end() &&
-          Next->getOpcode() == AMDGPU::V_NOP_e32) {
-        InsertPt = ++Next;
-        SkippedInitialUnclausedVmemPrologue = true;
-        continue;
-      }
-    }
-    break;
+    Candidates.push_back(CandBB);
+  }
+
+  // Record each candidate's current layout offset. This is the order in which
+  // the assembler emits blocks, and is used to determine the code range
+  // covered by a prefetch.
+  DenseMap<MachineBasicBlock *, uint64_t> CandidateOffsets;
+  uint64_t CodeSize = 0;
+  for (MachineBasicBlock &MB : MF) {
+    CodeSize = alignTo(CodeSize, MB.getAlignment());
+    if (isLoopFreeEntryPostDominator(MB, EntryBB))
+      CandidateOffsets[&MB] = CodeSize;
+    CodeSize += SIProgramInfo::getMachineBasicBlockCodeSize(MB, *TII);
   }
 
-  const SIInstrInfo *TII = ST.getInstrInfo();
   DebugLoc DL;
 
   // Each prefetch can transfer 4KiB of instructions. Retain the existing
@@ -128,24 +188,46 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
   unsigned NumPrefetches =
       llvm::divideCeil(ProgramSize + PrefetchSlack, BytesPerPrefetch);
   NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
-  for (unsigned I = 0; I < NumPrefetches; ++I) {
-    BuildMI(EntryBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
-        .addImm(0)                 // offset (placeholder, fixed up later)
-        .addReg(AMDGPU::SGPR_NULL) // soffset
-        .addImm(I);                // sdata (slot index, fixed up later)
+
+  size_t NumCandidates = Candidates.size();
+  unsigned Prefetches = 0;
+  // In each candidate block, prefetch as much code as necessary before control
+  // flow reaches the next candidate block, plus some slack to account for
+  // prefetch latency.
+  for (size_t Cand = 0, NextCand = 1; Cand < NumCandidates;
+       ++Cand, ++NextCand) {
+    unsigned PrefetchBeforeNext = NumPrefetches;
+    if (NextCand < NumCandidates) {
+      // To the offset of the next candidate we add:
+      // - PrefetchSlack: To make sure the last prefetch has the correct number
+      //   of cache lines.
+      // - BytesPerPrefetch: To account for the latency of the prefetch.
+      unsigned PrefetchesBeforeNext = llvm::divideCeil(
+          CandidateOffsets.lookup(Candidates[NextCand]) + PrefetchSlack +
+              BytesPerPrefetch,
+          BytesPerPrefetch);
+      PrefetchBeforeNext = std::min(PrefetchesBeforeNext, PrefetchBeforeNext);
+    }
+    MachineBasicBlock *CandBB = Candidates[Cand];
+    MachineBasicBlock::iterator InsertPt =
+        findMBBInsertionPoint(*CandBB, ST, CandBB == &EntryBB);
+    for (; Prefetches < PrefetchBeforeNext; ++Prefetches) {
+      BuildMI(*CandBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
+          .addImm(0)                 // offset (placeholder, fixed up later)
+          .addReg(AMDGPU::SGPR_NULL) // soffset
+          .addImm(Prefetches);       // sdata (slot index, fixed up later)
+    }
   }
 
-  // The instruction is fire-and-forget: it updates no wave wait counter and
-  // returns neither a result nor an error. It is therefore safe to insert
-  // after the wait-counter and hazard passes.
   MFI->setHasICachePrefetch(true);
   return true;
 }
 
-PreservedAnalyses
-llvm::AMDGPUInsertICachePrefetchPass::run(MachineFunction &MF,
-                                          MachineFunctionAnalysisManager &) {
-  if (!AMDGPUInsertICachePrefetch().run(MF))
+PreservedAnalyses llvm::AMDGPUInsertICachePrefetchPass::run(
+    MachineFunction &MF, MachineFunctionAnalysisManager &MFAM) {
+  auto &MLI = MFAM.getResult<MachineLoopAnalysis>(MF);
+  auto &PDT = MFAM.getResult<MachinePostDominatorTreeAnalysis>(MF);
+  if (!AMDGPUInsertICachePrefetch(MLI, PDT).run(MF))
     return PreservedAnalyses::all();
   auto PA = getMachineFunctionPassPreservedAnalyses();
   PA.preserveSet<CFGAnalyses>();
@@ -155,5 +237,9 @@ llvm::AMDGPUInsertICachePrefetchPass::run(MachineFunction &MF,
 char AMDGPUInsertICachePrefetchLegacy::ID = 0;
 char &llvm::AMDGPUInsertICachePrefetchID = AMDGPUInsertICachePrefetchLegacy::ID;
 
-INITIALIZE_PASS(AMDGPUInsertICachePrefetchLegacy, DEBUG_TYPE,
-                "AMDGPU Insert ICache Prefetch", false, false)
+INITIALIZE_PASS_BEGIN(AMDGPUInsertICachePrefetchLegacy, DEBUG_TYPE,
+                      "AMDGPU Insert ICache Prefetch", false, false)
+INITIALIZE_PASS_DEPENDENCY(MachineLoopInfoWrapperPass)
+INITIALIZE_PASS_DEPENDENCY(MachinePostDominatorTreeWrapperPass)
+INITIALIZE_PASS_END(AMDGPUInsertICachePrefetchLegacy, DEBUG_TYPE,
+                    "AMDGPU Insert ICache Prefetch", false, false)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
index be97f392d9bf4..7e8f53bee9db6 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
@@ -480,10 +480,10 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
         // The sdata operand contains the slot index [0, N) set by the pass.
         int64_t SlotIndex = TmpInst.getOperand(SdataIdx).getImm();
 
-        // If this is the first slot, emit the prefetch block start symbol
-        // before the instruction.
-        if (SlotIndex == 0)
-          OutStreamer->emitLabel(getPrefetchBlockStartSym());
+        // Emit a symbol for each prefetch instruction to calculate the offset.
+        MCSymbol *InstOffsetSym =
+            createTempSymbol("pref_inst_offset_" + Twine(SlotIndex));
+        OutStreamer->emitLabel(InstOffsetSym);
 
         // Create MCExpr for code size using label subtraction.
         // This gives the exact code size at assembly time.
@@ -495,18 +495,18 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
         const MCExpr *SlotIndexExpr =
             MCConstantExpr::create(SlotIndex, OutContext);
 
-        // Create MCExpr for the offset of the first prefetch instruction in the
+        // Create an MCExpr for this prefetch instruction's offset in the
         // function.
-        const MCExpr *PrefetchBlockOffset = MCBinaryExpr::createSub(
-            MCSymbolRefExpr::create(getPrefetchBlockStartSym(), OutContext),
+        const MCExpr *PrefetchInstOffset = MCBinaryExpr::createSub(
+            MCSymbolRefExpr::create(InstOffsetSym, OutContext),
             MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
 
         // Create MCExprs that will be evaluated at fixup time when symbol
         // positions are known.
         const MCExpr *CachelinesExpr = AMDGPUMCExpr::createPrefetchCachelines(
-            SlotIndexExpr, CodeSizeExpr, PrefetchBlockOffset, OutContext);
+            SlotIndexExpr, CodeSizeExpr, PrefetchInstOffset, OutContext);
         const MCExpr *OffsetExpr = AMDGPUMCExpr::createPrefetchOffset(
-            SlotIndexExpr, CodeSizeExpr, PrefetchBlockOffset, OutContext);
+            SlotIndexExpr, CodeSizeExpr, PrefetchInstOffset, OutContext);
 
         // Replace the offset and sdata operands with MCExprs.
         TmpInst.getOperand(OffsetIdx) = MCOperand::createExpr(OffsetExpr);
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index d349a2f29af11..f0bbab45f6e06 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -253,19 +253,9 @@ bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
   return true;
 }
 
-static uint64_t calcFirstPrefetchOffset(uint64_t PrefetchBlockOffset) {
-  constexpr unsigned CacheLineSize = 128;
-  // Start prefetching on the first cache line after the first prefetch
-  // instruction.
-  return llvm::alignDown(PrefetchBlockOffset, CacheLineSize) + CacheLineSize;
-}
-
-static uint64_t calcPrefetchSize(uint64_t CodeSizeInBytes,
-                                 uint64_t PrefetchBlockOffset) {
+static uint64_t calcPrefetchSize(uint64_t CodeSizeInBytes) {
   constexpr uint64_t MaxPrefetchSize = 64 * 1024; // 64KB ICache
-  uint64_t FirstPrefetch = calcFirstPrefetchOffset(PrefetchBlockOffset);
-  uint64_t ClampedCodeSize = std::min(CodeSizeInBytes, MaxPrefetchSize);
-  return ClampedCodeSize > FirstPrefetch ? ClampedCodeSize - FirstPrefetch : 0;
+  return std::min(CodeSizeInBytes, MaxPrefetchSize);
 }
 
 static uint64_t calcPrefetchSlotOffset(uint64_t SlotIndex) {
@@ -278,18 +268,16 @@ static uint64_t calcPrefetchSlotOffset(uint64_t SlotIndex) {
 
 bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
                                               const MCAssembler *Asm) const {
-  uint64_t SlotIndex = 0, CodeSizeInBytes = 0, PrefetchBlockOffset = 0;
-  if (!evaluateMCExprs(Args, Asm,
-                       {SlotIndex, CodeSizeInBytes, PrefetchBlockOffset}))
+  uint64_t SlotIndex = 0, CodeSizeInBytes = 0, InstOffset = 0;
+  if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes, InstOffset}))
     return false;
 
   // Constants for prefetch calculation.
   // Each instruction can prefetch up to 32 cachelines (5-bit sdata field, plus
   // one added). Cacheline size is 128 bytes. Each slot covers 4KiB.
-  constexpr unsigned MaxCachelinesPerPrefetch = 32;
+  constexpr uint64_t MaxCachelinesPerPrefetch = 32;
   constexpr unsigned CacheLineSize = 128;
-  uint64_t PrefetchSize =
-      calcPrefetchSize(CodeSizeInBytes, PrefetchBlockOffset);
+  uint64_t PrefetchSize = calcPrefetchSize(CodeSizeInBytes);
 
   // Calculate the byte offset for this slot.
   uint64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
@@ -307,8 +295,8 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
 
   // Calculate cachelines needed, clamped to max per instruction.
   uint64_t CachelinesNeeded = divideCeil(RemainingBytes, CacheLineSize);
-  uint64_t CachelineCount = std::min(
-      CachelinesNeeded, static_cast<uint64_t>(MaxCachelinesPerPrefetch));
+  uint64_t CachelineCount =
+      std::min(CachelinesNeeded, MaxCachelinesPerPrefetch);
 
   // The instruction adds 1 to the encoded sdata, so deduct it here.
   Res = MCValue::get(static_cast<int64_t>(CachelineCount - 1));
@@ -317,18 +305,13 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
 
 bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
                                           const MCAssembler *Asm) const {
-  uint64_t SlotIndex = 0, CodeSizeInBytes = 0, PrefetchBlockOffset = 0;
-  if (!evaluateMCExprs(Args, Asm,
-                       {SlotIndex, CodeSizeInBytes, PrefetchBlockOffset}))
+  uint64_t SlotIndex = 0, CodeSizeInBytes = 0, InstOffset = 0;
+  if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes, InstOffset}))
     return false;
 
-  constexpr uint64_t PrefetchInstSize = 8;
-
-  uint64_t FirstPrefetch = calcFirstPrefetchOffset(PrefetchBlockOffset);
-  uint64_t PrefetchSize =
-      calcPrefetchSize(CodeSizeInBytes, PrefetchBlockOffset);
+  int64_t PrefetchSize = calcPrefetchSize(CodeSizeInBytes);
 
-  uint64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
+  int64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
   if (SlotOffset >= PrefetchSize) {
     // Instruction semantics adds one to sdata when calculating the length of
     // the prefetch. This means that even a prefetch instruction with sdata == 0
@@ -339,10 +322,9 @@ bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
     return true;
   }
   // Prefetch is relative to this prefetch instruction's PC.
-  uint64_t PC = PrefetchBlockOffset + (SlotIndex * PrefetchInstSize);
-  uint64_t Target = FirstPrefetch + SlotOffset;
-  uint64_t Offset = Target - PC;
-  Res = MCValue::get(static_cast<int64_t>(Offset));
+  int64_t Offset =
+      static_cast<int64_t>(SlotOffset) - static_cast<int64_t>(InstOffset);
+  Res = MCValue::get(Offset);
   return true;
 }
 
@@ -456,16 +438,17 @@ AMDGPUMCExpr::createInstPrefSize(const MCExpr *CodeSizeBytes, MCContext &Ctx) {
 
 const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchCachelines(
     const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
-    const MCExpr *PrefetchBlockOffset, MCContext &Ctx) {
-  return create(AGVK_PrefetchCachelines,
-                {SlotIndex, CodeSizeBytes, PrefetchBlockOffset}, Ctx);
+    const MCExpr *InstOffset, MCContext &Ctx) {
+  return create(AGVK_PrefetchCachelines, {SlotIndex, CodeSizeBytes, InstOffset},
+                Ctx);
 }
 
-const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchOffset(
-    const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
-    const MCExpr *PrefetchBlockOffset, MCContext &Ctx) {
-  return create(AGVK_PrefetchOffset,
-                {SlotIndex, CodeSizeBytes, PrefetchBlockOffset}, Ctx);
+const AMDGPUMCExpr *
+AMDGPUMCExpr::createPrefetchOffset(const MCExpr *SlotIndex,
+                                   const MCExpr *CodeSizeBytes,
+                                   const MCExpr *InstOffset, MCContext &Ctx) {
+  return create(AGVK_PrefetchOffset, {SlotIndex, CodeSizeBytes, InstOffset},
+                Ctx);
 }
 
 const AMDGPUMCExpr *AMDGPUMCExpr::createLit(LitModifier Lit, int64_t Value,
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index 0db84e172f2ed..bba0e8b01ab3c 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -124,24 +124,25 @@ class AMDGPUMCExpr : public MCTargetExpr {
   /// slot.
   /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
   /// CodeSizeBytes is the total code size in bytes.
-  /// PrefetchBlockOffset is the byte offset of the first prefetch instruction
-  /// from the function entry.
+  /// InstOffset is the byte offset of this prefetch instruction from the
+  /// function entry.
   /// Returns the requested cacheline count minus one, encoded for the 5-bit
   /// sdata field.
   static const AMDGPUMCExpr *
   createPrefetchCachelines(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
-                           const MCExpr *PrefetchBlockOffset, MCContext &Ctx);
+                           const MCExpr *InstOffset, MCContext &Ctx);
 
   /// Create an expression for computing the byte offset for a prefetch slot.
   /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
   /// CodeSizeBytes is the total code size in bytes.
-  /// PrefetchBlockOffset is the byte offset of the first prefetch instruction
-  /// from the function entry.
+  /// InstOffset is the byte offset of this prefetch instruction from the
+  /// function entry.
   /// Returns the byte offset from the PC of the corresponding prefetch
   /// instruction to where this slot should start prefetching.
-  static const AMDGPUMCExpr *
-  createPrefetchOffset(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
-                       const MCExpr *PrefetchBlockOffset, MCContext &Ctx);
+  static const AMDGPUMCExpr *createPrefetchOffset(const MCExpr *SlotIndex,
+                                                  const MCExpr *CodeSizeBytes,
+                                                  const MCExpr *InstOffset,
+                                                  MCContext &Ctx);
 
   static const AMDGPUMCExpr *createLit(LitModifier Lit, int64_t Value,
                                        MCContext &Ctx);
diff --git a/llvm/lib/Target/AMDGPU/SIProgramInfo.cpp b/llvm/lib/Target/AMDGPU/SIProgramInfo.cpp
index 713f214bf315a..f0f80f24086c6 100644
--- a/llvm/lib/Target/AMDGPU/SIProgramInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIProgramInfo.cpp
@@ -226,17 +226,26 @@ uint64_t SIProgramInfo::getFunctionCodeSize(const MachineFunction &MF) {
 
   for (const MachineBasicBlock &MBB : MF) {
     CodeSize = alignTo(CodeSize, MBB.getAlignment());
+    CodeSize += getMachineBasicBlockCodeSize(MBB, *TII);
+  }
+
+  CodeSizeInBytes = CodeSize;
+  return CodeSize;
+}
+
+uint64_t
+SIProgramInfo::getMachineBasicBlockCodeSize(const MachineBasicBlock &MBB,
+                                            const SIInstrInfo &TII) {
+  uint64_t CodeSize = 0;
 
-    for (const MachineInstr &MI : MBB) {
-      // TODO: CodeSize should account for multiple functions.
+  for (const MachineInstr &MI : MBB) {
+    // TODO: CodeSize should account for multiple functions.
 
-      if (MI.isMetaInstruction())
-        continue;
+    if (MI.isMetaInstruction())
+      continue;
 
-      CodeSize += TII->getInstSizeInBytes(MI);
-    }
+    CodeSize += TII.getInstSizeInBytes(MI);
   }
 
-  CodeSizeInBytes = CodeSize;
   return CodeSize;
 }
diff --git a/llvm/lib/Target/AMDGPU/SIProgramInfo.h b/llvm/lib/Target/AMDGPU/SIProgramInfo.h
index fb56ebf88c96f..493a332013ed3 100644
--- a/llvm/lib/Target/AMDGPU/SIProgramInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIProgramInfo.h
@@ -26,7 +26,9 @@ namespace llvm {
 class GCNSubtarget;
 class MCContext;
 class MCExpr;
+class MachineBasicBlock;
 class MachineFunction;
+class SIInstrInfo;
 
 /// Track resource usage for kernels / entry functions.
 struct LLVM_EXTERNAL_VISIBILITY SIProgramInfo {
@@ -107,6 +109,10 @@ struct LLVM_EXTERNAL_VISIBILITY SIProgramInfo {
   // Get function code size and cache the value.
   uint64_t getFunctionCodeSize(const MachineFunction &MF);
 
+  // Get machine basic block code size.
+  static uint64_t getMachineBasicBlockCodeSize(const MachineBasicBlock &MBB,
+                                               const SIInstrInfo &TII);
+
   /// Compute the value of the ComputePGMRsrc1 register.
   const MCExpr *getComputePGMRSrc1(const GCNSubtarget &ST,
                                    MCContext &Ctx) const;
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 18777c160cc4e..879daea84f3a3 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -41,16 +41,16 @@ define amdgpu_kernel void @below_threshold() {
 
 ; Exercise several full 4 KiB prefetch slots and a partial final slot.
 ; GFX1250-OBJ-LABEL:      <partial_final_slot>:
-; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x68, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x1060, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2058, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3050, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4048, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5040, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6038, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7030, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8028, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9020, null, 24
+; GFX1250-OBJ:            s_prefetch_inst_pc_rel -0x18, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xfe0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x1fd8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2fd0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3fc8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4fc0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5fb8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6fb0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7fa8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8fa0, null, 25
 ; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x0, null, 0
 define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; GFX1250-LABEL: partial_final_slot:
@@ -58,18 +58,28 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT:  .Lpref_block_start0:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(0, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(1, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(2, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(3, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(4, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(5, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(6, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(7, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(8, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(9, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot), null, prefetchcachelines(10, .Lpref_func_end0-partial_final_slot, .Lpref_block_start0-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset_00:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_00-partial_final_slot), null, prefetchcachelines(0, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_00-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset_10:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_10-partial_final_slot), null, prefetchcachelines(1, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_10-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset_20:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_20-partial_final_slot), null, prefetchcachelines(2, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_20-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset_30:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_30-partial_final_slot), null, prefetchcachelines(3, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_30-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset_40:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_40-partial_final_slot), null, prefetchcachelines(4, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_40-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset_50:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_50-partial_final_slot), null, prefetchcachelines(5, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_50-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset_60:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_60-partial_final_slot), null, prefetchcachelines(6, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_60-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset_70:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_70-partial_final_slot), null, prefetchcachelines(7, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_70-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset_80:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_80-partial_final_slot), null, prefetchcachelines(8, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_80-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset_90:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_90-partial_final_slot), null, prefetchcachelines(9, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_90-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset_100:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_100-partial_final_slot), null, prefetchcachelines(10, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_100-partial_final_slot)
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
 ; GFX1250-NEXT:    v_mov_b32_e32 v0, 0
 ; GFX1250-NEXT:    ;;#ASMSTART
@@ -98,48 +108,62 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
   ret void
 }
 
-; Exercise the 16-instruction limit and reserve the cache line containing the
-; first prefetch instruction instead of attempting to replace all 64 KiB.
+; Exercise the 16-instruction limit for the complete 64 KiB ICache range.
 ; GFX1250-OBJ-LABEL:      <cache_size_limit>:
-; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x68, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x1060, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2058, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3050, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4048, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5040, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6038, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7030, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8028, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9020, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xa018, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xb010, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xc008, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xd000, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xdff8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xeff0, null, 30
+; GFX1250-OBJ:            s_prefetch_inst_pc_rel -0x18, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xfe0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x1fd8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2fd0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3fc8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4fc0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5fb8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6fb0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7fa8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8fa0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9f98, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xaf90, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xbf88, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xcf80, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xdf78, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xef70, null, 31
 define amdgpu_kernel void @cache_size_limit() {
 ; GFX1250-LABEL: cache_size_limit:
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT:  .Lpref_block_start1:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(0, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(1, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(2, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(3, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(4, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(5, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(6, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(7, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(8, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(9, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(10, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(11, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(12, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(13, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(14, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit), null, prefetchcachelines(15, .Lpref_func_end1-cache_size_limit, .Lpref_block_start1-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_01:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_01-cache_size_limit), null, prefetchcachelines(0, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_01-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_11:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_11-cache_size_limit), null, prefetchcachelines(1, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_11-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_21:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_21-cache_size_limit), null, prefetchcachelines(2, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_21-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_31:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_31-cache_size_limit), null, prefetchcachelines(3, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_31-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_41:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_41-cache_size_limit), null, prefetchcachelines(4, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_41-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_51:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_51-cache_size_limit), null, prefetchcachelines(5, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_51-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_61:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_61-cache_size_limit), null, prefetchcachelines(6, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_61-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_71:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_71-cache_size_limit), null, prefetchcachelines(7, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_71-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_81:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_81-cache_size_limit), null, prefetchcachelines(8, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_81-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_91:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_91-cache_size_limit), null, prefetchcachelines(9, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_91-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_101:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_101-cache_size_limit), null, prefetchcachelines(10, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_101-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_110:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_110-cache_size_limit), null, prefetchcachelines(11, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_110-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_120:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_120-cache_size_limit), null, prefetchcachelines(12, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_120-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_130:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_130-cache_size_limit), null, prefetchcachelines(13, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_130-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_140:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_140-cache_size_limit), null, prefetchcachelines(14, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_140-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset_150:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_150-cache_size_limit), null, prefetchcachelines(15, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_150-cache_size_limit)
 ; GFX1250-NEXT:    ;;#ASMSTART
 ; GFX1250-NEXT:    .space 65536
 ; GFX1250-NEXT:    ;;#ASMEND
@@ -159,6 +183,125 @@ define amdgpu_kernel void @cache_size_limit() {
   ret void
 }
 
+; The shared exit block post-dominates the entry block. Prefetches for code
+; beyond the branch arms are inserted there, rather than all in the entry.
+declare i32 @llvm.amdgcn.workgroup.id.x()
+
+define amdgpu_kernel void @postdominated_prefetch() {
+; GFX1250-LABEL: postdominated_prefetch:
+; GFX1250:       ; %bb.0: ; %entry
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:  .Lpref_inst_offset_02:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_02-postdominated_prefetch), null, prefetchcachelines(0, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_02-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset_12:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_12-postdominated_prefetch), null, prefetchcachelines(1, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_12-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset_22:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_22-postdominated_prefetch), null, prefetchcachelines(2, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_22-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset_32:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_32-postdominated_prefetch), null, prefetchcachelines(3, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_32-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset_42:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch), null, prefetchcachelines(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset_52:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch), null, prefetchcachelines(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch)
+; GFX1250-NEXT:    s_bfe_u32 s0, ttmp6, 0x4000c
+; GFX1250-NEXT:    s_and_b32 s1, ttmp6, 15
+; GFX1250-NEXT:    s_add_co_i32 s0, s0, 1
+; GFX1250-NEXT:    s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
+; GFX1250-NEXT:    s_mul_i32 s0, ttmp9, s0
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    s_add_co_i32 s1, s1, s0
+; GFX1250-NEXT:    s_cmp_eq_u32 s2, 0
+; GFX1250-NEXT:    s_cselect_b32 s0, ttmp9, s1
+; GFX1250-NEXT:    s_cmp_lg_u32 s0, 0
+; GFX1250-NEXT:    s_mov_b32 s0, 0
+; GFX1250-NEXT:    s_cbranch_scc0 .LBB3_4
+; GFX1250-NEXT:  ; %bb.1: ; %else
+; GFX1250-NEXT:    ;;#ASMSTART
+; GFX1250-NEXT:    .space 8000
+; GFX1250-NEXT:    ;;#ASMEND
+; GFX1250-NEXT:    s_and_not1_b32 vcc_lo, exec_lo, s0
+; GFX1250-NEXT:    s_cbranch_vccnz .LBB3_3
+; GFX1250-NEXT:  .LBB3_2: ; %then
+; GFX1250-NEXT:    ;;#ASMSTART
+; GFX1250-NEXT:    .space 8000
+; GFX1250-NEXT:    ;;#ASMEND
+; GFX1250-NEXT:  .LBB3_3: ; %join
+; GFX1250-NEXT:  .Lpref_inst_offset_62:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_62-postdominated_prefetch), null, prefetchcachelines(6, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_62-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset_72:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_72-postdominated_prefetch), null, prefetchcachelines(7, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_72-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset_82:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_82-postdominated_prefetch), null, prefetchcachelines(8, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_82-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset_92:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_92-postdominated_prefetch), null, prefetchcachelines(9, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_92-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset_102:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_102-postdominated_prefetch), null, prefetchcachelines(10, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_102-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset_111:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_111-postdominated_prefetch), null, prefetchcachelines(11, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_111-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset_121:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_121-postdominated_prefetch), null, prefetchcachelines(12, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_121-postdominated_prefetch)
+; GFX1250-NEXT:    ;;#ASMSTART
+; GFX1250-NEXT:    .space 32000
+; GFX1250-NEXT:    ;;#ASMEND
+; GFX1250-NEXT:    s_endpgm
+; GFX1250-NEXT:  .LBB3_4:
+; GFX1250-NEXT:    s_branch .LBB3_2
+; GFX1250-NEXT:  .Lpref_func_end2:
+;
+; NO-PREFETCH-LABEL: postdominated_prefetch:
+; NO-PREFETCH:       ; %bb.0: ; %entry
+; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
+; NO-PREFETCH-NEXT:    v_nop
+; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT:    s_bfe_u32 s0, ttmp6, 0x4000c
+; NO-PREFETCH-NEXT:    s_and_b32 s1, ttmp6, 15
+; NO-PREFETCH-NEXT:    s_add_co_i32 s0, s0, 1
+; NO-PREFETCH-NEXT:    s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
+; NO-PREFETCH-NEXT:    s_mul_i32 s0, ttmp9, s0
+; NO-PREFETCH-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; NO-PREFETCH-NEXT:    s_add_co_i32 s1, s1, s0
+; NO-PREFETCH-NEXT:    s_cmp_eq_u32 s2, 0
+; NO-PREFETCH-NEXT:    s_cselect_b32 s0, ttmp9, s1
+; NO-PREFETCH-NEXT:    s_cmp_lg_u32 s0, 0
+; NO-PREFETCH-NEXT:    s_mov_b32 s0, 0
+; NO-PREFETCH-NEXT:    s_cbranch_scc0 .LBB3_4
+; NO-PREFETCH-NEXT:  ; %bb.1: ; %else
+; NO-PREFETCH-NEXT:    ;;#ASMSTART
+; NO-PREFETCH-NEXT:    .space 8000
+; NO-PREFETCH-NEXT:    ;;#ASMEND
+; NO-PREFETCH-NEXT:    s_and_not1_b32 vcc_lo, exec_lo, s0
+; NO-PREFETCH-NEXT:    s_cbranch_vccnz .LBB3_3
+; NO-PREFETCH-NEXT:  .LBB3_2: ; %then
+; NO-PREFETCH-NEXT:    ;;#ASMSTART
+; NO-PREFETCH-NEXT:    .space 8000
+; NO-PREFETCH-NEXT:    ;;#ASMEND
+; NO-PREFETCH-NEXT:  .LBB3_3: ; %join
+; NO-PREFETCH-NEXT:    ;;#ASMSTART
+; NO-PREFETCH-NEXT:    .space 32000
+; NO-PREFETCH-NEXT:    ;;#ASMEND
+; NO-PREFETCH-NEXT:    s_endpgm
+; NO-PREFETCH-NEXT:  .LBB3_4:
+; NO-PREFETCH-NEXT:    s_branch .LBB3_2
+entry:
+  %id = call i32 @llvm.amdgcn.workgroup.id.x()
+  %cond = icmp eq i32 %id, 0
+  br i1 %cond, label %then, label %else
+
+then:
+  call void asm sideeffect ".space 8000", ""()
+  br label %join
+
+else:
+  call void asm sideeffect ".space 8000", ""()
+  br label %join
+
+join:
+  call void asm sideeffect ".space 32000", ""()
+  ret void
+}
+
 ; Non-entry functions must not get explicit prefetch instructions regardless
 ; of their size.
 define void @called_function() {

>From 279b50fbfcf752b76acfd1223241f4a96e7f9cb1 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 7 Aug 2026 06:55:07 -0500
Subject: [PATCH 10/22] Add CPop and update tests

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 .../AMDGPU/AMDGPUInsertICachePrefetch.cpp     |  3 +-
 llvm/test/CodeGen/AMDGPU/icache-prefetch.ll   | 50 +++++++++++--------
 llvm/test/CodeGen/AMDGPU/llc-pipeline.ll      |  4 ++
 3 files changed, 36 insertions(+), 21 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index be8703c4c9d6c..b331026feed9c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -215,7 +215,8 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
       BuildMI(*CandBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
           .addImm(0)                 // offset (placeholder, fixed up later)
           .addReg(AMDGPU::SGPR_NULL) // soffset
-          .addImm(Prefetches);       // sdata (slot index, fixed up later)
+          .addImm(Prefetches)        // sdata (slot index, fixed up later)
+          .addImm(0);                // cpol
     }
   }
 
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 879daea84f3a3..48f0a31e0dbda 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -183,8 +183,8 @@ define amdgpu_kernel void @cache_size_limit() {
   ret void
 }
 
-; The shared exit block post-dominates the entry block. Prefetches for code
-; beyond the branch arms are inserted there, rather than all in the entry.
+; The post-dominator chain distributes prefetches among the entry, Flow, and
+; shared exit blocks, rather than placing them all in the entry.
 declare i32 @llvm.amdgcn.workgroup.id.x()
 
 define amdgpu_kernel void @postdominated_prefetch() {
@@ -201,10 +201,6 @@ define amdgpu_kernel void @postdominated_prefetch() {
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_22-postdominated_prefetch), null, prefetchcachelines(2, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_22-postdominated_prefetch)
 ; GFX1250-NEXT:  .Lpref_inst_offset_32:
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_32-postdominated_prefetch), null, prefetchcachelines(3, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_32-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset_42:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch), null, prefetchcachelines(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset_52:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch), null, prefetchcachelines(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch)
 ; GFX1250-NEXT:    s_bfe_u32 s0, ttmp6, 0x4000c
 ; GFX1250-NEXT:    s_and_b32 s1, ttmp6, 15
 ; GFX1250-NEXT:    s_add_co_i32 s0, s0, 1
@@ -216,18 +212,29 @@ define amdgpu_kernel void @postdominated_prefetch() {
 ; GFX1250-NEXT:    s_cselect_b32 s0, ttmp9, s1
 ; GFX1250-NEXT:    s_cmp_lg_u32 s0, 0
 ; GFX1250-NEXT:    s_mov_b32 s0, 0
-; GFX1250-NEXT:    s_cbranch_scc0 .LBB3_4
+; GFX1250-NEXT:    s_cbranch_scc0 .LBB3_2
 ; GFX1250-NEXT:  ; %bb.1: ; %else
 ; GFX1250-NEXT:    ;;#ASMSTART
 ; GFX1250-NEXT:    .space 8000
 ; GFX1250-NEXT:    ;;#ASMEND
-; GFX1250-NEXT:    s_and_not1_b32 vcc_lo, exec_lo, s0
-; GFX1250-NEXT:    s_cbranch_vccnz .LBB3_3
-; GFX1250-NEXT:  .LBB3_2: ; %then
+; GFX1250-NEXT:    s_branch .LBB3_3
+; GFX1250-NEXT:  .LBB3_2:
+; GFX1250-NEXT:    s_mov_b32 s0, -1
+; GFX1250-NEXT:  .LBB3_3: ; %Flow
+; GFX1250-NEXT:  .Lpref_inst_offset_42:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch), null, prefetchcachelines(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset_52:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch), null, prefetchcachelines(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch)
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    s_and_b32 s0, s0, exec_lo
+; GFX1250-NEXT:    s_cselect_b32 s0, 1, 0
+; GFX1250-NEXT:    s_cmp_lg_u32 s0, 1
+; GFX1250-NEXT:    s_cbranch_scc1 .LBB3_5
+; GFX1250-NEXT:  ; %bb.4: ; %then
 ; GFX1250-NEXT:    ;;#ASMSTART
 ; GFX1250-NEXT:    .space 8000
 ; GFX1250-NEXT:    ;;#ASMEND
-; GFX1250-NEXT:  .LBB3_3: ; %join
+; GFX1250-NEXT:  .LBB3_5: ; %join
 ; GFX1250-NEXT:  .Lpref_inst_offset_62:
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_62-postdominated_prefetch), null, prefetchcachelines(6, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_62-postdominated_prefetch)
 ; GFX1250-NEXT:  .Lpref_inst_offset_72:
@@ -246,8 +253,6 @@ define amdgpu_kernel void @postdominated_prefetch() {
 ; GFX1250-NEXT:    .space 32000
 ; GFX1250-NEXT:    ;;#ASMEND
 ; GFX1250-NEXT:    s_endpgm
-; GFX1250-NEXT:  .LBB3_4:
-; GFX1250-NEXT:    s_branch .LBB3_2
 ; GFX1250-NEXT:  .Lpref_func_end2:
 ;
 ; NO-PREFETCH-LABEL: postdominated_prefetch:
@@ -266,24 +271,29 @@ define amdgpu_kernel void @postdominated_prefetch() {
 ; NO-PREFETCH-NEXT:    s_cselect_b32 s0, ttmp9, s1
 ; NO-PREFETCH-NEXT:    s_cmp_lg_u32 s0, 0
 ; NO-PREFETCH-NEXT:    s_mov_b32 s0, 0
-; NO-PREFETCH-NEXT:    s_cbranch_scc0 .LBB3_4
+; NO-PREFETCH-NEXT:    s_cbranch_scc0 .LBB3_2
 ; NO-PREFETCH-NEXT:  ; %bb.1: ; %else
 ; NO-PREFETCH-NEXT:    ;;#ASMSTART
 ; NO-PREFETCH-NEXT:    .space 8000
 ; NO-PREFETCH-NEXT:    ;;#ASMEND
-; NO-PREFETCH-NEXT:    s_and_not1_b32 vcc_lo, exec_lo, s0
-; NO-PREFETCH-NEXT:    s_cbranch_vccnz .LBB3_3
-; NO-PREFETCH-NEXT:  .LBB3_2: ; %then
+; NO-PREFETCH-NEXT:    s_branch .LBB3_3
+; NO-PREFETCH-NEXT:  .LBB3_2:
+; NO-PREFETCH-NEXT:    s_mov_b32 s0, -1
+; NO-PREFETCH-NEXT:  .LBB3_3: ; %Flow
+; NO-PREFETCH-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; NO-PREFETCH-NEXT:    s_and_b32 s0, s0, exec_lo
+; NO-PREFETCH-NEXT:    s_cselect_b32 s0, 1, 0
+; NO-PREFETCH-NEXT:    s_cmp_lg_u32 s0, 1
+; NO-PREFETCH-NEXT:    s_cbranch_scc1 .LBB3_5
+; NO-PREFETCH-NEXT:  ; %bb.4: ; %then
 ; NO-PREFETCH-NEXT:    ;;#ASMSTART
 ; NO-PREFETCH-NEXT:    .space 8000
 ; NO-PREFETCH-NEXT:    ;;#ASMEND
-; NO-PREFETCH-NEXT:  .LBB3_3: ; %join
+; NO-PREFETCH-NEXT:  .LBB3_5: ; %join
 ; NO-PREFETCH-NEXT:    ;;#ASMSTART
 ; NO-PREFETCH-NEXT:    .space 32000
 ; NO-PREFETCH-NEXT:    ;;#ASMEND
 ; NO-PREFETCH-NEXT:    s_endpgm
-; NO-PREFETCH-NEXT:  .LBB3_4:
-; NO-PREFETCH-NEXT:    s_branch .LBB3_2
 entry:
   %id = call i32 @llvm.amdgcn.workgroup.id.x()
   %cond = icmp eq i32 %id, 0
diff --git a/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll b/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
index 8f88d84280b47..9dd47db0287a9 100644
--- a/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
+++ b/llvm/test/CodeGen/AMDGPU/llc-pipeline.ll
@@ -455,6 +455,7 @@
 ; GCN-O1-NEXT:        AMDGPU Insert waits for SGPR read hazards
 ; GCN-O1-NEXT:        AMDGPU Lower VGPR Encoding
 ; GCN-O1-NEXT:        AMDGPU Insert Delay ALU
+; GCN-O1-NEXT:        MachinePostDominator Tree Construction
 ; GCN-O1-NEXT:        AMDGPU Insert ICache Prefetch
 ; GCN-O1-NEXT:        Branch relaxation pass
 ; GCN-O1-NEXT:        Register Usage Information Collector Pass
@@ -786,6 +787,7 @@
 ; GCN-O1-OPTS-NEXT:        AMDGPU Insert waits for SGPR read hazards
 ; GCN-O1-OPTS-NEXT:        AMDGPU Lower VGPR Encoding
 ; GCN-O1-OPTS-NEXT:        AMDGPU Insert Delay ALU
+; GCN-O1-OPTS-NEXT:        MachinePostDominator Tree Construction
 ; GCN-O1-OPTS-NEXT:        AMDGPU Insert ICache Prefetch
 ; GCN-O1-OPTS-NEXT:        Branch relaxation pass
 ; GCN-O1-OPTS-NEXT:        Register Usage Information Collector Pass
@@ -1122,6 +1124,7 @@
 ; GCN-O2-NEXT:        AMDGPU Insert waits for SGPR read hazards
 ; GCN-O2-NEXT:        AMDGPU Lower VGPR Encoding
 ; GCN-O2-NEXT:        AMDGPU Insert Delay ALU
+; GCN-O2-NEXT:        MachinePostDominator Tree Construction
 ; GCN-O2-NEXT:        AMDGPU Insert ICache Prefetch
 ; GCN-O2-NEXT:        Branch relaxation pass
 ; GCN-O2-NEXT:        Register Usage Information Collector Pass
@@ -1473,6 +1476,7 @@
 ; GCN-O3-NEXT:        AMDGPU Insert waits for SGPR read hazards
 ; GCN-O3-NEXT:        AMDGPU Lower VGPR Encoding
 ; GCN-O3-NEXT:        AMDGPU Insert Delay ALU
+; GCN-O3-NEXT:        MachinePostDominator Tree Construction
 ; GCN-O3-NEXT:        AMDGPU Insert ICache Prefetch
 ; GCN-O3-NEXT:        Branch relaxation pass
 ; GCN-O3-NEXT:        Register Usage Information Collector Pass

>From 8967681f2b1dbd1d3384b5c95e178d58ac0489da Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 7 Aug 2026 09:07:46 -0500
Subject: [PATCH 11/22] Revert remaining changes to SIPreEmitPeephole

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp | 13 ++++++-------
 1 file changed, 6 insertions(+), 7 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index e5b1de7bc4ec9..9b67cdd6be6f4 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -23,7 +23,6 @@
 #include "llvm/ADT/Statistic.h"
 #include "llvm/CodeGen/MachineDominators.h"
 #include "llvm/CodeGen/MachineFunctionPass.h"
-#include "llvm/CodeGen/MachineInstrBuilder.h"
 #include "llvm/CodeGen/MachineLoopInfo.h"
 #include "llvm/CodeGen/TargetSchedule.h"
 #include "llvm/Support/BranchProbability.h"
@@ -48,7 +47,6 @@ struct ModeFieldState {
 
 class SIPreEmitPeephole {
 private:
-  const GCNSubtarget *ST = nullptr;
   const SIInstrInfo *TII = nullptr;
   const SIRegisterInfo *TRI = nullptr;
   MachineLoopInfo *MLI = nullptr;
@@ -193,7 +191,8 @@ bool SIPreEmitPeephole::optimizeVccBranch(MachineInstr &MI) const {
 
   bool Changed = false;
   MachineBasicBlock &MBB = *MI.getParent();
-  const bool IsWave32 = ST->isWave32();
+  const GCNSubtarget &ST = MBB.getParent()->getSubtarget<GCNSubtarget>();
+  const bool IsWave32 = ST.isWave32();
   const unsigned CondReg = TRI->getVCC();
   const unsigned ExecReg = IsWave32 ? AMDGPU::EXEC_LO : AMDGPU::EXEC;
   const unsigned And = IsWave32 ? AMDGPU::S_AND_B32 : AMDGPU::S_AND_B64;
@@ -867,8 +866,8 @@ llvm::SIPreEmitPeepholePass::run(MachineFunction &MF,
 }
 
 bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
-  ST = &MF.getSubtarget<GCNSubtarget>();
-  TII = ST->getInstrInfo();
+  const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
+  TII = ST.getInstrInfo();
   TRI = &TII->getRegisterInfo();
   MLI = LoopInfo;
   bool Changed = false;
@@ -893,7 +892,7 @@ bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
       }
     }
 
-    if (!ST->hasVGPRIndexMode())
+    if (!ST.hasVGPRIndexMode())
       continue;
 
     MachineInstr *SetGPRMI = nullptr;
@@ -930,7 +929,7 @@ bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
   // side effects.
 
   // Perform the extra MF scans only for supported archs
-  if (!ST->hasGFX940Insts())
+  if (!ST.hasGFX940Insts())
     return Changed;
   for (MachineBasicBlock &MBB : MF) {
     // Unpack packed instructions overlapped by MFMAs. This allows the

>From 5778fd0e4b36084e92193aaaaba98a6d67b73483 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 7 Aug 2026 09:09:06 -0500
Subject: [PATCH 12/22] Code formatting

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 .../AMDGPU/AMDGPUInsertICachePrefetch.cpp     | 20 +++++++++----------
 1 file changed, 9 insertions(+), 11 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index b331026feed9c..dfcc9d2206322 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -83,9 +83,9 @@ class AMDGPUInsertICachePrefetchLegacy : public MachineFunctionPass {
 
 } // end anonymous namespace
 
-static MachineBasicBlock::iterator
-findMBBInsertionPoint(MachineBasicBlock &MBB, const GCNSubtarget &ST,
-                      bool IsEntryBlock) {
+static MachineBasicBlock::iterator findMBBInsertionPoint(MachineBasicBlock &MBB,
+                                                         const GCNSubtarget &ST,
+                                                         bool IsEntryBlock) {
   MachineBasicBlock::iterator InsertPt = MBB.begin();
 
   // Skip past any instructions that must remain at the very beginning:
@@ -99,8 +99,7 @@ findMBBInsertionPoint(MachineBasicBlock &MBB, const GCNSubtarget &ST,
       !IsEntryBlock || !ST.hasRequiresInitialUnclausedVmem();
   while (InsertPt != MBB.end()) {
     if (InsertPt->isDebugValue() || InsertPt->isCFIInstruction() ||
-        (IsEntryBlock &&
-         InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32)) {
+        (IsEntryBlock && InsertPt->getOpcode() == AMDGPU::S_SETREG_IMM32_B32)) {
       ++InsertPt;
       continue;
     }
@@ -156,8 +155,7 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
     MachineBasicBlock *CandBB = Node->getBlock();
     if (!CandBB)
       break;
-    if (CandBB == &EntryBB ||
-        !isLoopFreeEntryPostDominator(*CandBB, EntryBB))
+    if (CandBB == &EntryBB || !isLoopFreeEntryPostDominator(*CandBB, EntryBB))
       continue;
     Candidates.push_back(CandBB);
   }
@@ -202,10 +200,10 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
       // - PrefetchSlack: To make sure the last prefetch has the correct number
       //   of cache lines.
       // - BytesPerPrefetch: To account for the latency of the prefetch.
-      unsigned PrefetchesBeforeNext = llvm::divideCeil(
-          CandidateOffsets.lookup(Candidates[NextCand]) + PrefetchSlack +
-              BytesPerPrefetch,
-          BytesPerPrefetch);
+      unsigned PrefetchesBeforeNext =
+          llvm::divideCeil(CandidateOffsets.lookup(Candidates[NextCand]) +
+                               PrefetchSlack + BytesPerPrefetch,
+                           BytesPerPrefetch);
       PrefetchBeforeNext = std::min(PrefetchesBeforeNext, PrefetchBeforeNext);
     }
     MachineBasicBlock *CandBB = Candidates[Cand];

>From 21066b816269857cd02c7398ffbcbaf5bb42d5a5 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 7 Aug 2026 09:26:37 -0500
Subject: [PATCH 13/22] Skip prefetch if basic block sections are used

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp | 7 +++++++
 1 file changed, 7 insertions(+)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index dfcc9d2206322..6738dbc0bed74 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -135,6 +135,13 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
   if (!MFI->isEntryFunction())
     return false;
 
+  // Basic block sections can be independently placed by the linker, so the
+  // function does not form the contiguous address range assumed by the
+  // prefetch offset and size calculations.
+  if (MF.hasBBSections() ||
+      MF.getTarget().getBBSectionsType() != BasicBlockSection::None)
+    return false;
+
   SIProgramInfo PI;
   uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
   // The kernel descriptor can specify an instruction prefetch size of up to 256

>From 72b3ffa6e3b82836cfa47e1d1a307031f279ab46 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 20 Aug 2026 08:44:34 -0500
Subject: [PATCH 14/22] Make I$ size and prefetch size configurable

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/AMDGPU.td              | 26 +++++++++++
 .../AMDGPU/AMDGPUInsertICachePrefetch.cpp     | 46 +++++++++++++++----
 llvm/lib/Target/AMDGPU/GCNSubtarget.h         | 11 +++++
 .../CodeGen/AMDGPU/icache-prefetch-config.ll  | 39 ++++++++++++++++
 4 files changed, 113 insertions(+), 9 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index aabc845b3e0a9..c174ef9fa39ec 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -461,6 +461,28 @@ class SubtargetFeatureInstCacheLineSize <int Value> : SubtargetFeature <
 def FeatureInstCacheLineSize64  : SubtargetFeatureInstCacheLineSize<64>;
 def FeatureInstCacheLineSize128 : SubtargetFeatureInstCacheLineSize<128>;
 
+class SubtargetFeatureInstCacheSize <int Value> : SubtargetFeature <
+  "instcachesize"#Value,
+  "InstCacheSize",
+  !cast<string>(Value),
+  "Instruction cache size in bytes."
+>;
+
+def FeatureInstCacheSize32768 : SubtargetFeatureInstCacheSize<32768>;
+def FeatureInstCacheSize65536 : SubtargetFeatureInstCacheSize<65536>;
+
+class SubtargetFeaturePreferredInstPrefSize <int Value> : SubtargetFeature <
+  "preferredinstprefsize"#Value,
+  "PreferredInstPrefSize",
+  !cast<string>(Value),
+  "Preferred instruction prefetch size in bytes."
+>;
+
+def FeaturePreferredInstPrefSize16384
+    : SubtargetFeaturePreferredInstPrefSize<16384>;
+def FeaturePreferredInstPrefSize32768
+    : SubtargetFeaturePreferredInstPrefSize<32768>;
+
 class SubtargetFeatureDataCacheLineSize <int Value> : SubtargetFeature <
   "datacachelinesize"#Value,
   "DataCacheLineSize",
@@ -2447,6 +2469,8 @@ def FeatureISAVersion12 : FeatureSet<
    FeatureCvtPkNormVOP2Insts,
    FeatureCvtPkNormVOP3Insts,
    FeatureSmemPrefetchInsts,
+   FeatureInstCacheSize32768,
+   FeaturePreferredInstPrefSize16384,
    FeatureNoF16PseudoScalarTransInlineConstants,
    FeatureRealTrue16Insts,
    FeatureMTBUFInsts,
@@ -2541,6 +2565,8 @@ def FeatureISAVersion12_50_Common : FeatureSet<
    FeatureAsyncLoadToLDSInsts,
    FeatureAsyncStoreFromLDSInsts,
    FeatureSmemPrefetchInsts,
+   FeatureInstCacheSize65536,
+   FeaturePreferredInstPrefSize32768,
    FeatureNoF16PseudoScalarTransInlineConstants,
    FeatureRealTrue16Insts,
    FeatureVMovB64Inst,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index 6738dbc0bed74..e44a1dc55a0f6 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -27,6 +27,7 @@
 #include "llvm/CodeGen/MachinePostDominators.h"
 #include "llvm/InitializePasses.h"
 #include "llvm/Support/CommandLine.h"
+#include "llvm/Support/ErrorHandling.h"
 #include "llvm/TargetParser/Triple.h"
 
 using namespace llvm;
@@ -38,6 +39,11 @@ static cl::opt<bool>
                          cl::desc("Insert ICache prefetch instructions"),
                          cl::init(true), cl::Hidden);
 
+static cl::opt<unsigned> ICachePrefetchSize(
+    "amdgpu-icache-prefetch-size",
+    cl::desc("Override the preferred instruction prefetch size in bytes"),
+    cl::init(0), cl::Hidden);
+
 namespace {
 
 class AMDGPUInsertICachePrefetch {
@@ -83,6 +89,27 @@ class AMDGPUInsertICachePrefetchLegacy : public MachineFunctionPass {
 
 } // end anonymous namespace
 
+static uint64_t getPreferredICachePrefetchSize(const GCNSubtarget &ST) {
+  assert(ST.hasInstPrefSize());
+
+  uint64_t PreferredSize = ST.getPreferredInstPrefSize();
+  if (!ICachePrefetchSize.getNumOccurrences())
+    return PreferredSize;
+
+  uint32_t Mask, Shift, Width, CacheLineSize;
+  ST.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
+  uint64_t MaxPrefetchSize = (uint64_t{1} << Width) * CacheLineSize;
+  if (ICachePrefetchSize == 0 ||
+      ICachePrefetchSize % CacheLineSize != 0 ||
+      ICachePrefetchSize > MaxPrefetchSize)
+    report_fatal_error(
+        Twine("-amdgpu-icache-prefetch-size must be a non-zero multiple of ") +
+        Twine(CacheLineSize) + " bytes not exceeding " +
+        Twine(MaxPrefetchSize) + " bytes for " + ST.getCPU());
+
+  return ICachePrefetchSize;
+}
+
 static MachineBasicBlock::iterator findMBBInsertionPoint(MachineBasicBlock &MBB,
                                                          const GCNSubtarget &ST,
                                                          bool IsEntryBlock) {
@@ -123,7 +150,7 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
     return false;
 
   const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
-  if (!ST.hasICachePrefetch())
+  if (!ST.hasSmemPrefetchInsts() || !ST.hasInstPrefSize())
     return false;
 
   // Only run for AMDHSA - this is where kernel descriptors are used and
@@ -142,14 +169,14 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
       MF.getTarget().getBBSectionsType() != BasicBlockSection::None)
     return false;
 
+  uint64_t ICacheSize = ST.getInstCacheSize();
+  uint64_t PreferredPrefetchSize = getPreferredICachePrefetchSize(ST);
+  if (ICacheSize == 0 || PreferredPrefetchSize == 0)
+    return false;
+
   SIProgramInfo PI;
   uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
-  // The kernel descriptor can specify an instruction prefetch size of up to 256
-  // in INST_PREF_SIZE. At a granularity of 128B, this equals 32KiB of
-  // instructions that can be prefetched without inserting explicit prefetch
-  // instructions.
-  constexpr uint64_t MaxKDPrefetch = 1u << 15;
-  if (ProgramSize <= MaxKDPrefetch)
+  if (ProgramSize <= PreferredPrefetchSize)
     return false;
 
   const SIInstrInfo *TII = ST.getInstrInfo();
@@ -188,8 +215,9 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
   constexpr uint64_t PrefetchSlack = 2 * 1024;
   constexpr uint64_t BytesPerPrefetch = 4 * 1024;
   // Each prefetch can transfer up to 32 cachelines of 128 bytes = 4KiB.
-  // 16 instructions cover 64KiB (the full ICache size).
-  constexpr unsigned MaxNumPrefetchInsts = 16;
+  unsigned MaxNumPrefetchInsts = ICacheSize / BytesPerPrefetch;
+  if (MaxNumPrefetchInsts == 0)
+    return false;
   unsigned NumPrefetches =
       llvm::divideCeil(ProgramSize + PrefetchSlack, BytesPerPrefetch);
   NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 28e57bcf6a1c0..647f2a07dc547 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -81,6 +81,11 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
   // Instruction cache line size in bytes; set from TableGen subtarget features.
   unsigned InstCacheLineSize = 0;
 
+  // Instruction cache and preferred prefetch sizes in bytes. A zero value
+  // means that the target does not provide the corresponding policy.
+  unsigned InstCacheSize = 0;
+  unsigned PreferredInstPrefSize = 0;
+
   // Data (VMEM) cache line size in bytes; set from TableGen subtarget features.
   unsigned DataCacheLineSize = 0;
 
@@ -205,6 +210,12 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
   /// Instruction cache line size in bytes (64 for pre-GFX11, 128 for GFX11+).
   unsigned getInstCacheLineSize() const { return InstCacheLineSize; }
 
+  unsigned getInstCacheSize() const { return InstCacheSize; }
+
+  unsigned getPreferredInstPrefSize() const {
+    return PreferredInstPrefSize;
+  }
+
   /// Data (VMEM) cache line size in bytes (128 for gfx12), has no use before
   /// GFX12.
   unsigned getDataCacheLineSize() const { return DataCacheLineSize; }
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
new file mode 100644
index 0000000000000..4a409697481f4
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
@@ -0,0 +1,39 @@
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -o - %s | FileCheck -check-prefix=GFX1200 %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -o - %s | FileCheck -check-prefix=GFX1250 %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-size=16384 -o - %s | FileCheck -check-prefix=OVERRIDE-ENABLE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -amdgpu-icache-prefetch-size=32768 -o - %s | FileCheck -check-prefix=OVERRIDE-DISABLE %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -amdgpu-icache-prefetch-size=32769 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-SIZE %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -amdgpu-icache-prefetch-size=16385 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-SIZE %s
+
+; GFX12 defaults to a 32KiB I-cache and a 16KiB preferred INST_PREF_SIZE.
+; MI450 (gfx1250) defaults to a 64KiB I-cache and a 32KiB preferred size.
+; This 20KiB function is between those two preferred sizes.
+define amdgpu_kernel void @size_between_defaults() {
+; GFX1200-LABEL: size_between_defaults:
+; GFX1200:       s_prefetch_inst_pc_rel
+;
+; GFX1250-LABEL: size_between_defaults:
+; GFX1250-NOT:   s_prefetch_inst_pc_rel
+; GFX1250:       s_endpgm
+;
+; OVERRIDE-ENABLE-LABEL: size_between_defaults:
+; OVERRIDE-ENABLE:       s_prefetch_inst_pc_rel
+;
+; OVERRIDE-DISABLE-LABEL: size_between_defaults:
+; OVERRIDE-DISABLE-NOT:   s_prefetch_inst_pc_rel
+; OVERRIDE-DISABLE:       s_endpgm
+  call void asm sideeffect ".space 20000", ""()
+  ret void
+}
+
+; The GFX12 cache-size feature limits explicit prefetches to eight 4KiB slots.
+define amdgpu_kernel void @gfx12_cache_size_limit() {
+; GFX1200-LABEL: gfx12_cache_size_limit:
+; GFX1200:       prefetchoffset(7,
+; GFX1200-NOT:   prefetchoffset(8,
+; GFX1200:       s_endpgm
+  call void asm sideeffect ".space 65536", ""()
+  ret void
+}
+
+; INVALID-SIZE: LLVM ERROR: -amdgpu-icache-prefetch-size must be a non-zero multiple of 128 bytes not exceeding 32768 bytes for gfx1200

>From e2e85a653d90b8bfb02c8f93bae0a94aea9497cd Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Fri, 21 Aug 2026 02:42:32 -0500
Subject: [PATCH 15/22] Combine shader prefetch and prefetch instructions

Combine the initial shader prefetch driven by the INST_PREF_SIZE field
with the insertion of explicit prefetch instructions. The shader
prefetch loads as many bytes as preferred (through option or subtarget
default), the remaining code up to the I$ size limit is loaded with
explicit prefetch instructions inserted into the kernel CFG.

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp   |  11 +-
 .../AMDGPU/AMDGPUInsertICachePrefetch.cpp     |  43 ++--
 llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp  |  20 +-
 .../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp      |  74 +++----
 .../Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h |  17 +-
 .../lib/Target/AMDGPU/SIMachineFunctionInfo.h |  15 +-
 .../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp    |   8 +
 llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h |   4 +
 .../CodeGen/AMDGPU/icache-prefetch-config.ll  |  13 +-
 llvm/test/CodeGen/AMDGPU/icache-prefetch.ll   | 209 +++++++-----------
 10 files changed, 192 insertions(+), 222 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index fcbc576364e3f..3776390d8deb2 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -264,18 +264,17 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
   // right after the function code, so (Lfunc_end - func_sym) gives the
   // exact function code size in bytes.
   //
-  // When ICache prefetch is enabled, we set INST_PREF_SIZE to 1 because
-  // the s_prefetch_inst instructions handle the actual prefetching.
+  // When explicit ICache prefetch is enabled, INST_PREF_SIZE prefetches the
+  // initial cache-line prefix and enables scalar prefetch for the first WGP
+  // wave.
   if (STM.hasInstPrefSize()) {
     uint32_t Mask, Shift, Width, CacheLineSize;
     STM.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
 
     const MCExpr *InstPrefSize;
     if (MFI.hasICachePrefetch()) {
-      // Set INST_PREF_SIZE to 1. This enables prefetch and the first wave on
-      // the WGP gets SCALAR_PREFETCH_EN set to 1, while the remaining waves
-      // receive 0. This enables the s_prefetch_inst for the first wave.
-      InstPrefSize = MCConstantExpr::create(1, Ctx);
+      InstPrefSize =
+          MCConstantExpr::create(MFI.getICachePrefetchLines() - 1, Ctx);
     } else {
       const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
           MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index e44a1dc55a0f6..6823b8fcb632c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -99,8 +99,7 @@ static uint64_t getPreferredICachePrefetchSize(const GCNSubtarget &ST) {
   uint32_t Mask, Shift, Width, CacheLineSize;
   ST.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
   uint64_t MaxPrefetchSize = (uint64_t{1} << Width) * CacheLineSize;
-  if (ICachePrefetchSize == 0 ||
-      ICachePrefetchSize % CacheLineSize != 0 ||
+  if (ICachePrefetchSize == 0 || ICachePrefetchSize % CacheLineSize != 0 ||
       ICachePrefetchSize > MaxPrefetchSize)
     report_fatal_error(
         Twine("-amdgpu-icache-prefetch-size must be a non-zero multiple of ") +
@@ -174,6 +173,10 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
   if (ICacheSize == 0 || PreferredPrefetchSize == 0)
     return false;
 
+  unsigned CacheLineSize = ST.getInstCacheLineSize();
+  unsigned ICacheLines = ICacheSize / CacheLineSize;
+  unsigned DescriptorPrefetchLines = PreferredPrefetchSize / CacheLineSize;
+
   SIProgramInfo PI;
   uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
   if (ProgramSize <= PreferredPrefetchSize)
@@ -215,11 +218,16 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
   constexpr uint64_t PrefetchSlack = 2 * 1024;
   constexpr uint64_t BytesPerPrefetch = 4 * 1024;
   // Each prefetch can transfer up to 32 cachelines of 128 bytes = 4KiB.
-  unsigned MaxNumPrefetchInsts = ICacheSize / BytesPerPrefetch;
+  constexpr unsigned CacheLinesPerPrefetch = 32;
+  unsigned MaxNumPrefetchInsts = llvm::divideCeil(
+      ICacheLines - DescriptorPrefetchLines, CacheLinesPerPrefetch);
   if (MaxNumPrefetchInsts == 0)
     return false;
-  unsigned NumPrefetches =
-      llvm::divideCeil(ProgramSize + PrefetchSlack, BytesPerPrefetch);
+
+  MFI->setICachePrefetchLines(DescriptorPrefetchLines);
+  uint64_t ProgramPrefetchSize = ProgramSize + PrefetchSlack;
+  unsigned NumPrefetches = llvm::divideCeil(
+      ProgramPrefetchSize - PreferredPrefetchSize, BytesPerPrefetch);
   NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
 
   size_t NumCandidates = Candidates.size();
@@ -229,31 +237,36 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
   // prefetch latency.
   for (size_t Cand = 0, NextCand = 1; Cand < NumCandidates;
        ++Cand, ++NextCand) {
-    unsigned PrefetchBeforeNext = NumPrefetches;
+    unsigned TargetPrefetchCount = NumPrefetches;
     if (NextCand < NumCandidates) {
       // To the offset of the next candidate we add:
       // - PrefetchSlack: To make sure the last prefetch has the correct number
       //   of cache lines.
       // - BytesPerPrefetch: To account for the latency of the prefetch.
-      unsigned PrefetchesBeforeNext =
-          llvm::divideCeil(CandidateOffsets.lookup(Candidates[NextCand]) +
-                               PrefetchSlack + BytesPerPrefetch,
-                           BytesPerPrefetch);
-      PrefetchBeforeNext = std::min(PrefetchesBeforeNext, PrefetchBeforeNext);
+      uint64_t CandidatePrefetchSize =
+          CandidateOffsets.lookup(Candidates[NextCand]) + PrefetchSlack +
+          BytesPerPrefetch;
+
+      if (CandidatePrefetchSize <= PreferredPrefetchSize)
+        continue;
+
+      unsigned PrefetchesBeforeNext = llvm::divideCeil(
+          CandidatePrefetchSize - PreferredPrefetchSize, BytesPerPrefetch);
+      TargetPrefetchCount = std::min(PrefetchesBeforeNext, TargetPrefetchCount);
     }
     MachineBasicBlock *CandBB = Candidates[Cand];
     MachineBasicBlock::iterator InsertPt =
         findMBBInsertionPoint(*CandBB, ST, CandBB == &EntryBB);
-    for (; Prefetches < PrefetchBeforeNext; ++Prefetches) {
+    for (; Prefetches < TargetPrefetchCount; ++Prefetches) {
       BuildMI(*CandBB, InsertPt, DL, TII->get(AMDGPU::S_PREFETCH_INST_PC_REL))
-          .addImm(0)                 // offset (placeholder, fixed up later)
+          .addImm(DescriptorPrefetchLines + Prefetches * CacheLinesPerPrefetch)
+          // Function-relative target cache-line index, fixed up later.
           .addReg(AMDGPU::SGPR_NULL) // soffset
-          .addImm(Prefetches)        // sdata (slot index, fixed up later)
+          .addImm(0)                 // sdata (fixed up later)
           .addImm(0);                // cpol
     }
   }
 
-  MFI->setHasICachePrefetch(true);
   return true;
 }
 
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
index 7e8f53bee9db6..6941b8f4a3a1d 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMCInstLower.cpp
@@ -468,8 +468,9 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
     MCInstLowering.lower(MI, TmpInst);
 
     // Fix up S_PREFETCH_INST_PC_REL instructions inserted by the ICache
-    // prefetch pass. Replace the slot index in the sdata operand with an
-    // MCExpr that computes the cacheline count based on exact code size.
+    // prefetch pass. The provisional offset operand holds a function-relative
+    // target cache-line index; replace it and sdata with expressions that use
+    // final code layout.
     if (MI->getOpcode() == AMDGPU::S_PREFETCH_INST_PC_REL) {
       const SIMachineFunctionInfo *MFI = MF->getInfo<SIMachineFunctionInfo>();
       if (MFI->hasICachePrefetch()) {
@@ -477,12 +478,10 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
         constexpr unsigned OffsetIdx = 0;
         constexpr unsigned SdataIdx = 2;
 
-        // The sdata operand contains the slot index [0, N) set by the pass.
-        int64_t SlotIndex = TmpInst.getOperand(SdataIdx).getImm();
+        int64_t TargetCacheLine = TmpInst.getOperand(OffsetIdx).getImm();
 
         // Emit a symbol for each prefetch instruction to calculate the offset.
-        MCSymbol *InstOffsetSym =
-            createTempSymbol("pref_inst_offset_" + Twine(SlotIndex));
+        MCSymbol *InstOffsetSym = createTempSymbol("pref_inst_offset");
         OutStreamer->emitLabel(InstOffsetSym);
 
         // Create MCExpr for code size using label subtraction.
@@ -491,9 +490,8 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
             MCSymbolRefExpr::create(getPrefetchEndSym(), OutContext),
             MCSymbolRefExpr::create(CurrentFnSym, OutContext), OutContext);
 
-        // Create MCExpr for the slot index.
-        const MCExpr *SlotIndexExpr =
-            MCConstantExpr::create(SlotIndex, OutContext);
+        const MCExpr *TargetCacheLineExpr =
+            MCConstantExpr::create(TargetCacheLine, OutContext);
 
         // Create an MCExpr for this prefetch instruction's offset in the
         // function.
@@ -504,9 +502,9 @@ void AMDGPUAsmPrinter::emitInstruction(const MachineInstr *MI) {
         // Create MCExprs that will be evaluated at fixup time when symbol
         // positions are known.
         const MCExpr *CachelinesExpr = AMDGPUMCExpr::createPrefetchCachelines(
-            SlotIndexExpr, CodeSizeExpr, PrefetchInstOffset, OutContext);
+            TargetCacheLineExpr, CodeSizeExpr, PrefetchInstOffset, OutContext);
         const MCExpr *OffsetExpr = AMDGPUMCExpr::createPrefetchOffset(
-            SlotIndexExpr, CodeSizeExpr, PrefetchInstOffset, OutContext);
+            TargetCacheLineExpr, CodeSizeExpr, PrefetchInstOffset, OutContext);
 
         // Replace the offset and sdata operands with MCExprs.
         TmpInst.getOperand(OffsetIdx) = MCOperand::createExpr(OffsetExpr);
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index f0bbab45f6e06..3fe152d831a31 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -253,50 +253,31 @@ bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
   return true;
 }
 
-static uint64_t calcPrefetchSize(uint64_t CodeSizeInBytes) {
-  constexpr uint64_t MaxPrefetchSize = 64 * 1024; // 64KB ICache
-  return std::min(CodeSizeInBytes, MaxPrefetchSize);
-}
-
-static uint64_t calcPrefetchSlotOffset(uint64_t SlotIndex) {
-  constexpr unsigned MaxCachelinesPerPrefetch = 32;
-  constexpr unsigned CacheLineSize = 128;
-  constexpr unsigned BytesPerPrefetch =
-      MaxCachelinesPerPrefetch * CacheLineSize;
-  return SlotIndex * BytesPerPrefetch;
-}
-
 bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
                                               const MCAssembler *Asm) const {
-  uint64_t SlotIndex = 0, CodeSizeInBytes = 0, InstOffset = 0;
-  if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes, InstOffset}))
+  uint64_t TargetCacheLine = 0, CodeSizeInBytes = 0, InstOffset = 0;
+  if (!evaluateMCExprs(Args, Asm,
+                       {TargetCacheLine, CodeSizeInBytes, InstOffset}))
     return false;
 
-  // Constants for prefetch calculation.
-  // Each instruction can prefetch up to 32 cachelines (5-bit sdata field, plus
-  // one added). Cacheline size is 128 bytes. Each slot covers 4KiB.
+  const MCSubtargetInfo *STI = Ctx.getSubtargetInfo();
   constexpr uint64_t MaxCachelinesPerPrefetch = 32;
-  constexpr unsigned CacheLineSize = 128;
-  uint64_t PrefetchSize = calcPrefetchSize(CodeSizeInBytes);
-
-  // Calculate the byte offset for this slot.
-  uint64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
+  unsigned CacheLineSize = AMDGPU::IsaInfo::getInstCacheLineSize(*STI);
+  uint64_t ICacheLines =
+      AMDGPU::IsaInfo::getInstCacheSize(*STI) / CacheLineSize;
+  uint64_t CodeSizeInLines = divideCeil(CodeSizeInBytes, CacheLineSize);
+  uint64_t PrefetchEnd = std::min(CodeSizeInLines, ICacheLines);
 
-  // If this slot starts beyond the prefetchable region, use the minimum
+  // If this target starts beyond the prefetchable region, use the minimum
   // encoded prefetch size. evaluatePrefetchOffset() targets the instruction's
   // own cache line for such slots.
-  if (SlotOffset >= PrefetchSize) {
+  if (TargetCacheLine >= PrefetchEnd) {
     Res = MCValue::get(static_cast<int64_t>(0));
     return true;
   }
 
-  // Calculate remaining bytes from this slot's offset.
-  uint64_t RemainingBytes = PrefetchSize - SlotOffset;
-
-  // Calculate cachelines needed, clamped to max per instruction.
-  uint64_t CachelinesNeeded = divideCeil(RemainingBytes, CacheLineSize);
   uint64_t CachelineCount =
-      std::min(CachelinesNeeded, MaxCachelinesPerPrefetch);
+      std::min(PrefetchEnd - TargetCacheLine, MaxCachelinesPerPrefetch);
 
   // The instruction adds 1 to the encoded sdata, so deduct it here.
   Res = MCValue::get(static_cast<int64_t>(CachelineCount - 1));
@@ -305,14 +286,17 @@ bool AMDGPUMCExpr::evaluatePrefetchCachelines(MCValue &Res,
 
 bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
                                           const MCAssembler *Asm) const {
-  uint64_t SlotIndex = 0, CodeSizeInBytes = 0, InstOffset = 0;
-  if (!evaluateMCExprs(Args, Asm, {SlotIndex, CodeSizeInBytes, InstOffset}))
+  uint64_t TargetCacheLine = 0, CodeSizeInBytes = 0, InstOffset = 0;
+  if (!evaluateMCExprs(Args, Asm,
+                       {TargetCacheLine, CodeSizeInBytes, InstOffset}))
     return false;
 
-  int64_t PrefetchSize = calcPrefetchSize(CodeSizeInBytes);
-
-  int64_t SlotOffset = calcPrefetchSlotOffset(SlotIndex);
-  if (SlotOffset >= PrefetchSize) {
+  const MCSubtargetInfo *STI = Ctx.getSubtargetInfo();
+  unsigned CacheLineSize = AMDGPU::IsaInfo::getInstCacheLineSize(*STI);
+  uint64_t ICacheLines =
+      AMDGPU::IsaInfo::getInstCacheSize(*STI) / CacheLineSize;
+  uint64_t CodeSizeInLines = divideCeil(CodeSizeInBytes, CacheLineSize);
+  if (TargetCacheLine >= std::min(CodeSizeInLines, ICacheLines)) {
     // Instruction semantics adds one to sdata when calculating the length of
     // the prefetch. This means that even a prefetch instruction with sdata == 0
     // still performs a prefetch. Therefore, to make this prefetch neutral, we
@@ -322,8 +306,8 @@ bool AMDGPUMCExpr::evaluatePrefetchOffset(MCValue &Res,
     return true;
   }
   // Prefetch is relative to this prefetch instruction's PC.
-  int64_t Offset =
-      static_cast<int64_t>(SlotOffset) - static_cast<int64_t>(InstOffset);
+  int64_t Offset = static_cast<int64_t>(TargetCacheLine * CacheLineSize) -
+                   static_cast<int64_t>(InstOffset);
   Res = MCValue::get(Offset);
   return true;
 }
@@ -437,18 +421,18 @@ AMDGPUMCExpr::createInstPrefSize(const MCExpr *CodeSizeBytes, MCContext &Ctx) {
 }
 
 const AMDGPUMCExpr *AMDGPUMCExpr::createPrefetchCachelines(
-    const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
+    const MCExpr *TargetCacheLine, const MCExpr *CodeSizeBytes,
     const MCExpr *InstOffset, MCContext &Ctx) {
-  return create(AGVK_PrefetchCachelines, {SlotIndex, CodeSizeBytes, InstOffset},
-                Ctx);
+  return create(AGVK_PrefetchCachelines,
+                {TargetCacheLine, CodeSizeBytes, InstOffset}, Ctx);
 }
 
 const AMDGPUMCExpr *
-AMDGPUMCExpr::createPrefetchOffset(const MCExpr *SlotIndex,
+AMDGPUMCExpr::createPrefetchOffset(const MCExpr *TargetCacheLine,
                                    const MCExpr *CodeSizeBytes,
                                    const MCExpr *InstOffset, MCContext &Ctx) {
-  return create(AGVK_PrefetchOffset, {SlotIndex, CodeSizeBytes, InstOffset},
-                Ctx);
+  return create(AGVK_PrefetchOffset,
+                {TargetCacheLine, CodeSizeBytes, InstOffset}, Ctx);
 }
 
 const AMDGPUMCExpr *AMDGPUMCExpr::createLit(LitModifier Lit, int64_t Value,
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
index bba0e8b01ab3c..0f9efa69eacee 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.h
@@ -120,26 +120,27 @@ class AMDGPUMCExpr : public MCTargetExpr {
   static const AMDGPUMCExpr *createInstPrefSize(const MCExpr *CodeSizeBytes,
                                                 MCContext &Ctx);
 
-  /// Create an expression for computing the encoded sdata field for a prefetch
-  /// slot.
-  /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
+  /// Create an expression for computing the encoded sdata field for a
+  /// prefetch target. TargetCacheLine is the function-relative target
+  /// cache-line index.
   /// CodeSizeBytes is the total code size in bytes.
   /// InstOffset is the byte offset of this prefetch instruction from the
   /// function entry.
   /// Returns the requested cacheline count minus one, encoded for the 5-bit
   /// sdata field.
   static const AMDGPUMCExpr *
-  createPrefetchCachelines(const MCExpr *SlotIndex, const MCExpr *CodeSizeBytes,
+  createPrefetchCachelines(const MCExpr *TargetCacheLine,
+                           const MCExpr *CodeSizeBytes,
                            const MCExpr *InstOffset, MCContext &Ctx);
 
-  /// Create an expression for computing the byte offset for a prefetch slot.
-  /// SlotIndex is the 0-based index of the prefetch instruction (0-15).
+  /// Create an expression for computing the byte offset for a prefetch target.
+  /// TargetCacheLine is the function-relative target cache-line index.
   /// CodeSizeBytes is the total code size in bytes.
   /// InstOffset is the byte offset of this prefetch instruction from the
   /// function entry.
   /// Returns the byte offset from the PC of the corresponding prefetch
-  /// instruction to where this slot should start prefetching.
-  static const AMDGPUMCExpr *createPrefetchOffset(const MCExpr *SlotIndex,
+  /// instruction to where this target should start prefetching.
+  static const AMDGPUMCExpr *createPrefetchOffset(const MCExpr *TargetCacheLine,
                                                   const MCExpr *CodeSizeBytes,
                                                   const MCExpr *InstOffset,
                                                   MCContext &Ctx);
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
index 4b85168bdfed6..78240ecd67e37 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
@@ -490,9 +490,10 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
   bool HasNonSpillStackObjects = false;
   bool IsStackRealigned = false;
 
-  // Set when ICache prefetch instructions have been inserted in the entry
-  // block. This tells AsmPrinter to set rsrc3 INST_PREF_SIZE to 0.
-  bool HasICachePrefetch = false;
+  // Number of cache lines prefetched through rsrc3 INST_PREF_SIZE when
+  // explicit ICache prefetch instructions are used. A value of zero means the
+  // function does not use the explicit-prefetch scheme.
+  unsigned ICachePrefetchLines = 0;
 
   unsigned NumSpilledSGPRs = 0;
   unsigned NumSpilledVGPRs = 0;
@@ -1130,10 +1131,12 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
     IsStackRealigned = Realigned;
   }
 
-  bool hasICachePrefetch() const { return HasICachePrefetch; }
+  bool hasICachePrefetch() const { return ICachePrefetchLines != 0; }
 
-  void setHasICachePrefetch(bool Prefetch = true) {
-    HasICachePrefetch = Prefetch;
+  unsigned getICachePrefetchLines() const { return ICachePrefetchLines; }
+
+  void setICachePrefetchLines(unsigned Lines) {
+    ICachePrefetchLines = Lines;
   }
 
   unsigned getNumSpilledSGPRs() const {
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index e45359719d6c9..32e1a1d7c780c 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1107,6 +1107,14 @@ unsigned getInstCacheLineSize(const MCSubtargetInfo &STI) {
   return 64;
 }
 
+unsigned getInstCacheSize(const MCSubtargetInfo &STI) {
+  if (STI.getFeatureBits().test(FeatureInstCacheSize65536))
+    return 64 * 1024;
+  if (STI.getFeatureBits().test(FeatureInstCacheSize32768))
+    return 32 * 1024;
+  return 0;
+}
+
 unsigned getWavefrontSize(const MCSubtargetInfo &STI) {
   if (STI.getFeatureBits().test(FeatureWavefrontSize16))
     return 16;
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index bce059a0c18a7..a427856cb691e 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -186,6 +186,10 @@ inline bool targetIDSettingsConflict(TargetIDSetting Lhs, TargetIDSetting Rhs) {
 /// \returns Instruction cache line size in bytes for given subtarget \p STI.
 unsigned getInstCacheLineSize(const MCSubtargetInfo &STI);
 
+/// \returns Instruction cache size in bytes for given subtarget \p STI, or
+/// zero if the subtarget does not provide an instruction-cache policy.
+unsigned getInstCacheSize(const MCSubtargetInfo &STI);
+
 /// \returns Wavefront size for given subtarget \p STI.
 unsigned getWavefrontSize(const MCSubtargetInfo &STI);
 
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
index 4a409697481f4..2ff4047c3d781 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
@@ -10,14 +10,16 @@
 ; This 20KiB function is between those two preferred sizes.
 define amdgpu_kernel void @size_between_defaults() {
 ; GFX1200-LABEL: size_between_defaults:
-; GFX1200:       s_prefetch_inst_pc_rel
+; GFX1200:       prefetchoffset(128,
+; GFX1200:       .amdhsa_inst_pref_size 127
 ;
 ; GFX1250-LABEL: size_between_defaults:
 ; GFX1250-NOT:   s_prefetch_inst_pc_rel
 ; GFX1250:       s_endpgm
 ;
 ; OVERRIDE-ENABLE-LABEL: size_between_defaults:
-; OVERRIDE-ENABLE:       s_prefetch_inst_pc_rel
+; OVERRIDE-ENABLE:       prefetchoffset(128,
+; OVERRIDE-ENABLE:       .amdhsa_inst_pref_size 127
 ;
 ; OVERRIDE-DISABLE-LABEL: size_between_defaults:
 ; OVERRIDE-DISABLE-NOT:   s_prefetch_inst_pc_rel
@@ -26,11 +28,12 @@ define amdgpu_kernel void @size_between_defaults() {
   ret void
 }
 
-; The GFX12 cache-size feature limits explicit prefetches to eight 4KiB slots.
+; The GFX12 cache-size feature limits explicit prefetches to the four 4KiB
+; slots remaining after the 16KiB descriptor prefetch.
 define amdgpu_kernel void @gfx12_cache_size_limit() {
 ; GFX1200-LABEL: gfx12_cache_size_limit:
-; GFX1200:       prefetchoffset(7,
-; GFX1200-NOT:   prefetchoffset(8,
+; GFX1200:       prefetchoffset(224,
+; GFX1200-NOT:   prefetchoffset(256,
 ; GFX1200:       s_endpgm
   call void asm sideeffect ".space 65536", ""()
   ret void
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 48f0a31e0dbda..7759aa017933d 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -5,6 +5,9 @@
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=false -o - %s | \
 ; RUN:   FileCheck -check-prefix=NO-PREFETCH %s
 
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -amdgpu-icache-prefetch-size=16384 -o - %s | \
+; RUN:   FileCheck -check-prefix=GFX1250-DIST %s
+
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
 ; RUN: llvm-objdump -d %t.o | FileCheck -check-prefix=GFX1250-OBJ %s
 
@@ -18,9 +21,10 @@
 define amdgpu_kernel void @below_threshold() {
 ; GFX1250-LABEL: below_threshold:
 ; GFX1250:       ; %bb.0:
-; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    ;;#ASMSTART
 ; GFX1250-NEXT:    .space 30000
 ; GFX1250-NEXT:    ;;#ASMEND
@@ -28,9 +32,10 @@ define amdgpu_kernel void @below_threshold() {
 ;
 ; NO-PREFETCH-LABEL: below_threshold:
 ; NO-PREFETCH:       ; %bb.0:
-; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; NO-PREFETCH-NEXT:    v_nop
 ; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT:    s_mov_b64 s[64:65], 0
+; NO-PREFETCH-NEXT:    v_nop
+; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; NO-PREFETCH-NEXT:    ;;#ASMSTART
 ; NO-PREFETCH-NEXT:    .space 30000
 ; NO-PREFETCH-NEXT:    ;;#ASMEND
@@ -39,47 +44,25 @@ define amdgpu_kernel void @below_threshold() {
   ret void
 }
 
-; Exercise several full 4 KiB prefetch slots and a partial final slot.
+; The descriptor prefetches the first 32KiB; explicit requests cover the
+; remaining code in the 64KiB I-cache window.
 ; GFX1250-OBJ-LABEL:      <partial_final_slot>:
-; GFX1250-OBJ:            s_prefetch_inst_pc_rel -0x18, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xfe0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x1fd8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2fd0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3fc8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4fc0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5fb8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6fb0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7fa8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8fa0, null, 25
+; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x7ff8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8ff0, null, 25
 ; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x0, null, 0
 define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; GFX1250-LABEL: partial_final_slot:
 ; GFX1250:       ; %bb.0:
-; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT:  .Lpref_inst_offset_00:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_00-partial_final_slot), null, prefetchcachelines(0, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_00-partial_final_slot)
-; GFX1250-NEXT:  .Lpref_inst_offset_10:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_10-partial_final_slot), null, prefetchcachelines(1, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_10-partial_final_slot)
-; GFX1250-NEXT:  .Lpref_inst_offset_20:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_20-partial_final_slot), null, prefetchcachelines(2, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_20-partial_final_slot)
-; GFX1250-NEXT:  .Lpref_inst_offset_30:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_30-partial_final_slot), null, prefetchcachelines(3, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_30-partial_final_slot)
-; GFX1250-NEXT:  .Lpref_inst_offset_40:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_40-partial_final_slot), null, prefetchcachelines(4, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_40-partial_final_slot)
-; GFX1250-NEXT:  .Lpref_inst_offset_50:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_50-partial_final_slot), null, prefetchcachelines(5, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_50-partial_final_slot)
-; GFX1250-NEXT:  .Lpref_inst_offset_60:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_60-partial_final_slot), null, prefetchcachelines(6, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_60-partial_final_slot)
-; GFX1250-NEXT:  .Lpref_inst_offset_70:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_70-partial_final_slot), null, prefetchcachelines(7, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_70-partial_final_slot)
-; GFX1250-NEXT:  .Lpref_inst_offset_80:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_80-partial_final_slot), null, prefetchcachelines(8, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_80-partial_final_slot)
-; GFX1250-NEXT:  .Lpref_inst_offset_90:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_90-partial_final_slot), null, prefetchcachelines(9, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_90-partial_final_slot)
-; GFX1250-NEXT:  .Lpref_inst_offset_100:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_100-partial_final_slot), null, prefetchcachelines(10, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset_100-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset0:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset1:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset2:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot)
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
 ; GFX1250-NEXT:    v_mov_b32_e32 v0, 0
 ; GFX1250-NEXT:    ;;#ASMSTART
@@ -92,9 +75,10 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ;
 ; NO-PREFETCH-LABEL: partial_final_slot:
 ; NO-PREFETCH:       ; %bb.0:
-; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; NO-PREFETCH-NEXT:    v_nop
 ; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT:    s_mov_b64 s[64:65], 0
+; NO-PREFETCH-NEXT:    v_nop
+; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; NO-PREFETCH-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
 ; NO-PREFETCH-NEXT:    v_mov_b32_e32 v0, 0
 ; NO-PREFETCH-NEXT:    ;;#ASMSTART
@@ -108,62 +92,40 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
   ret void
 }
 
-; Exercise the 16-instruction limit for the complete 64 KiB ICache range.
+; The 32KiB descriptor prefix leaves eight explicit requests for the complete
+; 64KiB I-cache range.
 ; GFX1250-OBJ-LABEL:      <cache_size_limit>:
-; GFX1250-OBJ:            s_prefetch_inst_pc_rel -0x18, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xfe0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x1fd8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2fd0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3fc8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4fc0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5fb8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6fb0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7fa8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8fa0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9f98, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xaf90, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xbf88, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xcf80, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xdf78, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xef70, null, 31
+; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x7ff8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8ff0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9fe8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xafe0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xbfd8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xcfd0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xdfc8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xefc0, null, 31
 define amdgpu_kernel void @cache_size_limit() {
 ; GFX1250-LABEL: cache_size_limit:
 ; GFX1250:       ; %bb.0:
-; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT:  .Lpref_inst_offset_01:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_01-cache_size_limit), null, prefetchcachelines(0, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_01-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_11:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_11-cache_size_limit), null, prefetchcachelines(1, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_11-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_21:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_21-cache_size_limit), null, prefetchcachelines(2, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_21-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_31:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_31-cache_size_limit), null, prefetchcachelines(3, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_31-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_41:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_41-cache_size_limit), null, prefetchcachelines(4, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_41-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_51:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_51-cache_size_limit), null, prefetchcachelines(5, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_51-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_61:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_61-cache_size_limit), null, prefetchcachelines(6, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_61-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_71:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_71-cache_size_limit), null, prefetchcachelines(7, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_71-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_81:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_81-cache_size_limit), null, prefetchcachelines(8, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_81-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_91:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_91-cache_size_limit), null, prefetchcachelines(9, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_91-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_101:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_101-cache_size_limit), null, prefetchcachelines(10, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_101-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_110:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_110-cache_size_limit), null, prefetchcachelines(11, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_110-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_120:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_120-cache_size_limit), null, prefetchcachelines(12, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_120-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_130:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(13, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_130-cache_size_limit), null, prefetchcachelines(13, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_130-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_140:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(14, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_140-cache_size_limit), null, prefetchcachelines(14, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_140-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset_150:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(15, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_150-cache_size_limit), null, prefetchcachelines(15, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset_150-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset3:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset3-cache_size_limit), null, prefetchcachelines(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset3-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset4:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset4-cache_size_limit), null, prefetchcachelines(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset4-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset5:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset5-cache_size_limit), null, prefetchcachelines(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset5-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset6:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset6-cache_size_limit), null, prefetchcachelines(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset6-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset7:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit), null, prefetchcachelines(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset8:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit), null, prefetchcachelines(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset9:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset10:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit)
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    ;;#ASMSTART
 ; GFX1250-NEXT:    .space 65536
 ; GFX1250-NEXT:    ;;#ASMEND
@@ -172,9 +134,10 @@ define amdgpu_kernel void @cache_size_limit() {
 ;
 ; NO-PREFETCH-LABEL: cache_size_limit:
 ; NO-PREFETCH:       ; %bb.0:
-; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; NO-PREFETCH-NEXT:    v_nop
 ; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT:    s_mov_b64 s[64:65], 0
+; NO-PREFETCH-NEXT:    v_nop
+; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; NO-PREFETCH-NEXT:    ;;#ASMSTART
 ; NO-PREFETCH-NEXT:    .space 65536
 ; NO-PREFETCH-NEXT:    ;;#ASMEND
@@ -183,24 +146,25 @@ define amdgpu_kernel void @cache_size_limit() {
   ret void
 }
 
-; The post-dominator chain distributes prefetches among the entry, Flow, and
-; shared exit blocks, rather than placing them all in the entry.
+; With the default 32KiB descriptor prefix, explicit prefetches are deferred
+; to the shared exit block.
 declare i32 @llvm.amdgcn.workgroup.id.x()
 
 define amdgpu_kernel void @postdominated_prefetch() {
+; GFX1250-DIST-LABEL: postdominated_prefetch:
+; GFX1250-DIST-NOT:   s_prefetch_inst_pc_rel
+; GFX1250-DIST:       .LBB3_3: ; %Flow
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset
+; GFX1250-DIST-NEXT:  s_prefetch_inst_pc_rel prefetchoffset(128,
+; GFX1250-DIST:       .LBB3_5: ; %join
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset
+; GFX1250-DIST-NEXT:  s_prefetch_inst_pc_rel prefetchoffset(192,
 ; GFX1250-LABEL: postdominated_prefetch:
 ; GFX1250:       ; %bb.0: ; %entry
-; GFX1250-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT:  .Lpref_inst_offset_02:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(0, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_02-postdominated_prefetch), null, prefetchcachelines(0, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_02-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset_12:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(1, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_12-postdominated_prefetch), null, prefetchcachelines(1, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_12-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset_22:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(2, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_22-postdominated_prefetch), null, prefetchcachelines(2, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_22-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset_32:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(3, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_32-postdominated_prefetch), null, prefetchcachelines(3, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_32-postdominated_prefetch)
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_bfe_u32 s0, ttmp6, 0x4000c
 ; GFX1250-NEXT:    s_and_b32 s1, ttmp6, 15
 ; GFX1250-NEXT:    s_add_co_i32 s0, s0, 1
@@ -221,10 +185,6 @@ define amdgpu_kernel void @postdominated_prefetch() {
 ; GFX1250-NEXT:  .LBB3_2:
 ; GFX1250-NEXT:    s_mov_b32 s0, -1
 ; GFX1250-NEXT:  .LBB3_3: ; %Flow
-; GFX1250-NEXT:  .Lpref_inst_offset_42:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch), null, prefetchcachelines(4, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_42-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset_52:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch), null, prefetchcachelines(5, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_52-postdominated_prefetch)
 ; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
 ; GFX1250-NEXT:    s_and_b32 s0, s0, exec_lo
 ; GFX1250-NEXT:    s_cselect_b32 s0, 1, 0
@@ -235,20 +195,16 @@ define amdgpu_kernel void @postdominated_prefetch() {
 ; GFX1250-NEXT:    .space 8000
 ; GFX1250-NEXT:    ;;#ASMEND
 ; GFX1250-NEXT:  .LBB3_5: ; %join
-; GFX1250-NEXT:  .Lpref_inst_offset_62:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(6, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_62-postdominated_prefetch), null, prefetchcachelines(6, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_62-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset_72:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(7, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_72-postdominated_prefetch), null, prefetchcachelines(7, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_72-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset_82:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(8, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_82-postdominated_prefetch), null, prefetchcachelines(8, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_82-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset_92:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(9, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_92-postdominated_prefetch), null, prefetchcachelines(9, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_92-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset_102:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(10, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_102-postdominated_prefetch), null, prefetchcachelines(10, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_102-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset_111:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(11, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_111-postdominated_prefetch), null, prefetchcachelines(11, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_111-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset_121:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(12, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_121-postdominated_prefetch), null, prefetchcachelines(12, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset_121-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset11:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset11-postdominated_prefetch), null, prefetchcachelines(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset11-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset12:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset12-postdominated_prefetch), null, prefetchcachelines(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset12-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset13:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset13-postdominated_prefetch), null, prefetchcachelines(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset13-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset14:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset14-postdominated_prefetch), null, prefetchcachelines(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset14-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset15:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset15-postdominated_prefetch), null, prefetchcachelines(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset15-postdominated_prefetch)
 ; GFX1250-NEXT:    ;;#ASMSTART
 ; GFX1250-NEXT:    .space 32000
 ; GFX1250-NEXT:    ;;#ASMEND
@@ -257,9 +213,10 @@ define amdgpu_kernel void @postdominated_prefetch() {
 ;
 ; NO-PREFETCH-LABEL: postdominated_prefetch:
 ; NO-PREFETCH:       ; %bb.0: ; %entry
-; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[0:1] scope:SCOPE_SE
-; NO-PREFETCH-NEXT:    v_nop
 ; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; NO-PREFETCH-NEXT:    s_mov_b64 s[64:65], 0
+; NO-PREFETCH-NEXT:    v_nop
+; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; NO-PREFETCH-NEXT:    s_bfe_u32 s0, ttmp6, 0x4000c
 ; NO-PREFETCH-NEXT:    s_and_b32 s1, ttmp6, 15
 ; NO-PREFETCH-NEXT:    s_add_co_i32 s0, s0, 1

>From a86eac524f7d13f9f109d80a019c394b02347bfa Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Wed, 30 Sep 2026 08:53:38 -0500
Subject: [PATCH 16/22] Make treshold and initial prefetch size independent

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/AMDGPU.td              |  20 +-
 .../AMDGPU/AMDGPUInsertICachePrefetch.cpp     |  71 +++--
 llvm/lib/Target/AMDGPU/GCNSubtarget.h         |   8 +-
 .../CodeGen/AMDGPU/icache-prefetch-config.ll  |  93 ++++--
 llvm/test/CodeGen/AMDGPU/icache-prefetch.ll   | 288 ++++++++++++++----
 5 files changed, 367 insertions(+), 113 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index c174ef9fa39ec..9e9cdd10d99cb 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -471,17 +471,17 @@ class SubtargetFeatureInstCacheSize <int Value> : SubtargetFeature <
 def FeatureInstCacheSize32768 : SubtargetFeatureInstCacheSize<32768>;
 def FeatureInstCacheSize65536 : SubtargetFeatureInstCacheSize<65536>;
 
-class SubtargetFeaturePreferredInstPrefSize <int Value> : SubtargetFeature <
-  "preferredinstprefsize"#Value,
-  "PreferredInstPrefSize",
+class SubtargetFeatureInitialInstPrefSize <int Value> : SubtargetFeature <
+  "initialinstprefsize"#Value,
+  "InitialInstPrefSize",
   !cast<string>(Value),
-  "Preferred instruction prefetch size in bytes."
+  "Initial instruction prefetch size in bytes."
 >;
 
-def FeaturePreferredInstPrefSize16384
-    : SubtargetFeaturePreferredInstPrefSize<16384>;
-def FeaturePreferredInstPrefSize32768
-    : SubtargetFeaturePreferredInstPrefSize<32768>;
+def FeatureInitialInstPrefSize8192
+    : SubtargetFeatureInitialInstPrefSize<8192>;
+def FeatureInitialInstPrefSize16384
+    : SubtargetFeatureInitialInstPrefSize<16384>;
 
 class SubtargetFeatureDataCacheLineSize <int Value> : SubtargetFeature <
   "datacachelinesize"#Value,
@@ -2470,7 +2470,7 @@ def FeatureISAVersion12 : FeatureSet<
    FeatureCvtPkNormVOP3Insts,
    FeatureSmemPrefetchInsts,
    FeatureInstCacheSize32768,
-   FeaturePreferredInstPrefSize16384,
+   FeatureInitialInstPrefSize16384,
    FeatureNoF16PseudoScalarTransInlineConstants,
    FeatureRealTrue16Insts,
    FeatureMTBUFInsts,
@@ -2566,7 +2566,7 @@ def FeatureISAVersion12_50_Common : FeatureSet<
    FeatureAsyncStoreFromLDSInsts,
    FeatureSmemPrefetchInsts,
    FeatureInstCacheSize65536,
-   FeaturePreferredInstPrefSize32768,
+   FeatureInitialInstPrefSize8192,
    FeatureNoF16PseudoScalarTransInlineConstants,
    FeatureRealTrue16Insts,
    FeatureVMovB64Inst,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index 6823b8fcb632c..73766a35e2ed2 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -39,9 +39,14 @@ static cl::opt<bool>
                          cl::desc("Insert ICache prefetch instructions"),
                          cl::init(true), cl::Hidden);
 
-static cl::opt<unsigned> ICachePrefetchSize(
-    "amdgpu-icache-prefetch-size",
-    cl::desc("Override the preferred instruction prefetch size in bytes"),
+static cl::opt<unsigned> ICachePrefetchInitialSize(
+    "amdgpu-icache-prefetch-initial-size",
+    cl::desc("Override the initial instruction prefetch size in bytes"),
+    cl::init(0), cl::Hidden);
+
+static cl::opt<unsigned> ICachePrefetchThreshold(
+    "amdgpu-icache-prefetch-threshold",
+    cl::desc("Override the explicit instruction prefetch threshold in bytes"),
     cl::init(0), cl::Hidden);
 
 namespace {
@@ -89,24 +94,48 @@ class AMDGPUInsertICachePrefetchLegacy : public MachineFunctionPass {
 
 } // end anonymous namespace
 
-static uint64_t getPreferredICachePrefetchSize(const GCNSubtarget &ST) {
-  assert(ST.hasInstPrefSize());
+struct ICachePrefetchConfig {
+  uint64_t InitialSize;
+  uint64_t Threshold;
+};
 
-  uint64_t PreferredSize = ST.getPreferredInstPrefSize();
-  if (!ICachePrefetchSize.getNumOccurrences())
-    return PreferredSize;
+static ICachePrefetchConfig getICachePrefetchConfig(const GCNSubtarget &ST) {
+  assert(ST.hasInstPrefSize());
 
   uint32_t Mask, Shift, Width, CacheLineSize;
   ST.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
-  uint64_t MaxPrefetchSize = (uint64_t{1} << Width) * CacheLineSize;
-  if (ICachePrefetchSize == 0 || ICachePrefetchSize % CacheLineSize != 0 ||
-      ICachePrefetchSize > MaxPrefetchSize)
+  uint64_t DescriptorPrefetchCapacity = (uint64_t{1} << Width) * CacheLineSize;
+  uint64_t InitialSize = ICachePrefetchInitialSize.getNumOccurrences()
+                             ? ICachePrefetchInitialSize
+                             : ST.getInitialInstPrefSize();
+  uint64_t Threshold = ICachePrefetchThreshold.getNumOccurrences()
+                           ? ICachePrefetchThreshold
+                           : DescriptorPrefetchCapacity;
+
+  if (InitialSize == 0 || InitialSize % CacheLineSize != 0 ||
+      InitialSize > DescriptorPrefetchCapacity)
     report_fatal_error(
-        Twine("-amdgpu-icache-prefetch-size must be a non-zero multiple of ") +
+        Twine("-amdgpu-icache-prefetch-initial-size must be a non-zero "
+              "multiple of ") +
         Twine(CacheLineSize) + " bytes not exceeding " +
-        Twine(MaxPrefetchSize) + " bytes for " + ST.getCPU());
+        Twine(DescriptorPrefetchCapacity) + " bytes for " + ST.getCPU());
+
+  uint64_t ICacheSize = ST.getInstCacheSize();
+  if (Threshold == 0 || Threshold % CacheLineSize != 0 ||
+      Threshold > ICacheSize)
+    report_fatal_error(
+        Twine("-amdgpu-icache-prefetch-threshold must be a non-zero multiple "
+              "of ") +
+        Twine(CacheLineSize) + " bytes not exceeding " + Twine(ICacheSize) +
+        " bytes for " + ST.getCPU());
+
+  if (InitialSize > Threshold)
+    report_fatal_error(
+        Twine("-amdgpu-icache-prefetch-initial-size must not exceed "
+              "-amdgpu-icache-prefetch-threshold for ") +
+        ST.getCPU());
 
-  return ICachePrefetchSize;
+  return {InitialSize, Threshold};
 }
 
 static MachineBasicBlock::iterator findMBBInsertionPoint(MachineBasicBlock &MBB,
@@ -169,17 +198,17 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
     return false;
 
   uint64_t ICacheSize = ST.getInstCacheSize();
-  uint64_t PreferredPrefetchSize = getPreferredICachePrefetchSize(ST);
-  if (ICacheSize == 0 || PreferredPrefetchSize == 0)
+  if (ICacheSize == 0)
     return false;
+  const ICachePrefetchConfig Config = getICachePrefetchConfig(ST);
 
   unsigned CacheLineSize = ST.getInstCacheLineSize();
   unsigned ICacheLines = ICacheSize / CacheLineSize;
-  unsigned DescriptorPrefetchLines = PreferredPrefetchSize / CacheLineSize;
+  unsigned DescriptorPrefetchLines = Config.InitialSize / CacheLineSize;
 
   SIProgramInfo PI;
   uint64_t ProgramSize = PI.getFunctionCodeSize(MF);
-  if (ProgramSize <= PreferredPrefetchSize)
+  if (ProgramSize <= Config.Threshold)
     return false;
 
   const SIInstrInfo *TII = ST.getInstrInfo();
@@ -227,7 +256,7 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
   MFI->setICachePrefetchLines(DescriptorPrefetchLines);
   uint64_t ProgramPrefetchSize = ProgramSize + PrefetchSlack;
   unsigned NumPrefetches = llvm::divideCeil(
-      ProgramPrefetchSize - PreferredPrefetchSize, BytesPerPrefetch);
+      ProgramPrefetchSize - Config.InitialSize, BytesPerPrefetch);
   NumPrefetches = std::min(MaxNumPrefetchInsts, NumPrefetches);
 
   size_t NumCandidates = Candidates.size();
@@ -247,11 +276,11 @@ bool AMDGPUInsertICachePrefetch::run(MachineFunction &MF) {
           CandidateOffsets.lookup(Candidates[NextCand]) + PrefetchSlack +
           BytesPerPrefetch;
 
-      if (CandidatePrefetchSize <= PreferredPrefetchSize)
+      if (CandidatePrefetchSize <= Config.InitialSize)
         continue;
 
       unsigned PrefetchesBeforeNext = llvm::divideCeil(
-          CandidatePrefetchSize - PreferredPrefetchSize, BytesPerPrefetch);
+          CandidatePrefetchSize - Config.InitialSize, BytesPerPrefetch);
       TargetPrefetchCount = std::min(PrefetchesBeforeNext, TargetPrefetchCount);
     }
     MachineBasicBlock *CandBB = Candidates[Cand];
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 647f2a07dc547..df1f651de13f0 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -81,10 +81,10 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
   // Instruction cache line size in bytes; set from TableGen subtarget features.
   unsigned InstCacheLineSize = 0;
 
-  // Instruction cache and preferred prefetch sizes in bytes. A zero value
+  // Instruction cache and initial prefetch sizes in bytes. A zero value
   // means that the target does not provide the corresponding policy.
   unsigned InstCacheSize = 0;
-  unsigned PreferredInstPrefSize = 0;
+  unsigned InitialInstPrefSize = 0;
 
   // Data (VMEM) cache line size in bytes; set from TableGen subtarget features.
   unsigned DataCacheLineSize = 0;
@@ -212,9 +212,7 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
 
   unsigned getInstCacheSize() const { return InstCacheSize; }
 
-  unsigned getPreferredInstPrefSize() const {
-    return PreferredInstPrefSize;
-  }
+  unsigned getInitialInstPrefSize() const { return InitialInstPrefSize; }
 
   /// Data (VMEM) cache line size in bytes (128 for gfx12), has no use before
   /// GFX12.
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
index 2ff4047c3d781..574ca53a26363 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
@@ -1,42 +1,83 @@
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -o - %s | FileCheck -check-prefix=GFX1200 %s
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -o - %s | FileCheck -check-prefix=GFX1250 %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-size=16384 -o - %s | FileCheck -check-prefix=OVERRIDE-ENABLE %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -amdgpu-icache-prefetch-size=32768 -o - %s | FileCheck -check-prefix=OVERRIDE-DISABLE %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -amdgpu-icache-prefetch-size=32769 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-SIZE %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -amdgpu-icache-prefetch-size=16385 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-SIZE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -o - %s | FileCheck -check-prefix=INITIAL-16K %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=49152 -o - %s | FileCheck -check-prefix=THRESHOLD-48K %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=THRESHOLD-16K %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=EQUAL-16K %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=0 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=8193 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=32896 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=0 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=32769 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=65664 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=8192 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-ORDER %s
 
-; GFX12 defaults to a 32KiB I-cache and a 16KiB preferred INST_PREF_SIZE.
-; MI450 (gfx1250) defaults to a 64KiB I-cache and a 32KiB preferred size.
-; This 20KiB function is between those two preferred sizes.
-define amdgpu_kernel void @size_between_defaults() {
-; GFX1200-LABEL: size_between_defaults:
-; GFX1200:       prefetchoffset(128,
+; The default threshold is the 32KiB descriptor capacity on both targets.
+; gfx1200 initially prefetches 16KiB, while gfx1250 initially prefetches 8KiB.
+define amdgpu_kernel void @above_default_threshold() {
+; GFX1200-LABEL: above_default_threshold:
+; GFX1200:       s_prefetch_inst_pc_rel prefetchoffset(128,
 ; GFX1200:       .amdhsa_inst_pref_size 127
 ;
-; GFX1250-LABEL: size_between_defaults:
+; GFX1250-LABEL: above_default_threshold:
+; GFX1250:       s_prefetch_inst_pc_rel prefetchoffset(64,
+; GFX1250:       .amdhsa_inst_pref_size 63
+;
+; INITIAL-16K-LABEL: above_default_threshold:
+; INITIAL-16K:       s_prefetch_inst_pc_rel prefetchoffset(128,
+; INITIAL-16K:       .amdhsa_inst_pref_size 127
+;
+; THRESHOLD-48K-LABEL: above_default_threshold:
+; THRESHOLD-48K-NOT:   s_prefetch_inst_pc_rel
+; THRESHOLD-48K:       s_endpgm
+; THRESHOLD-48K:       .amdhsa_inst_pref_size ((instprefsize(
+  call void asm sideeffect ".space 40000", ""()
+  ret void
+}
+
+; The gfx1250 prologue and s_endpgm add 32 bytes, making these functions
+; exactly 32KiB and four bytes over 32KiB respectively.
+define amdgpu_kernel void @at_default_threshold() {
+; GFX1250-LABEL: at_default_threshold:
 ; GFX1250-NOT:   s_prefetch_inst_pc_rel
 ; GFX1250:       s_endpgm
+  call void asm sideeffect ".space 32736", ""()
+  ret void
+}
+
+define amdgpu_kernel void @above_default_threshold_by_four() {
+; GFX1250-LABEL: above_default_threshold_by_four:
+; GFX1250:       s_prefetch_inst_pc_rel prefetchoffset(64,
+  call void asm sideeffect ".space 32740", ""()
+  ret void
+}
+
+; Lowering only the threshold activates explicit prefetching while retaining
+; the default 8KiB initial descriptor coverage. Setting initial == threshold
+; is also valid and changes the descriptor coverage independently.
+define amdgpu_kernel void @between_override_thresholds() {
+; THRESHOLD-16K-LABEL: between_override_thresholds:
+; THRESHOLD-16K:       s_prefetch_inst_pc_rel prefetchoffset(64,
+; THRESHOLD-16K:       .amdhsa_inst_pref_size 63
 ;
-; OVERRIDE-ENABLE-LABEL: size_between_defaults:
-; OVERRIDE-ENABLE:       prefetchoffset(128,
-; OVERRIDE-ENABLE:       .amdhsa_inst_pref_size 127
-;
-; OVERRIDE-DISABLE-LABEL: size_between_defaults:
-; OVERRIDE-DISABLE-NOT:   s_prefetch_inst_pc_rel
-; OVERRIDE-DISABLE:       s_endpgm
+; EQUAL-16K-LABEL: between_override_thresholds:
+; EQUAL-16K:       s_prefetch_inst_pc_rel prefetchoffset(128,
+; EQUAL-16K:       .amdhsa_inst_pref_size 127
   call void asm sideeffect ".space 20000", ""()
   ret void
 }
 
-; The GFX12 cache-size feature limits explicit prefetches to the four 4KiB
-; slots remaining after the 16KiB descriptor prefetch.
-define amdgpu_kernel void @gfx12_cache_size_limit() {
-; GFX1200-LABEL: gfx12_cache_size_limit:
-; GFX1200:       prefetchoffset(224,
-; GFX1200-NOT:   prefetchoffset(256,
-; GFX1200:       s_endpgm
+; INITIAL-16K-LABEL: cache_size_limit:
+; INITIAL-16K-COUNT-12: s_prefetch_inst_pc_rel
+; INITIAL-16K-NEXT:     s_mov_b64
+; GFX1250-LABEL:     cache_size_limit:
+; GFX1250-COUNT-14:  s_prefetch_inst_pc_rel
+; GFX1250-NEXT:      s_mov_b64
+define amdgpu_kernel void @cache_size_limit() {
   call void asm sideeffect ".space 65536", ""()
   ret void
 }
 
-; INVALID-SIZE: LLVM ERROR: -amdgpu-icache-prefetch-size must be a non-zero multiple of 128 bytes not exceeding 32768 bytes for gfx1200
+; INVALID-INITIAL: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must be a non-zero multiple of 128 bytes not exceeding 32768 bytes for gfx1250
+; INVALID-THRESHOLD: LLVM ERROR: -amdgpu-icache-prefetch-threshold must be a non-zero multiple of 128 bytes not exceeding 65536 bytes for gfx1250
+; INVALID-ORDER: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must not exceed -amdgpu-icache-prefetch-threshold for gfx1250
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 7759aa017933d..286cfb8cb5949 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -5,7 +5,7 @@
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=false -o - %s | \
 ; RUN:   FileCheck -check-prefix=NO-PREFETCH %s
 
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -amdgpu-icache-prefetch-size=16384 -o - %s | \
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -amdgpu-icache-prefetch-initial-size=16384 -o - %s | \
 ; RUN:   FileCheck -check-prefix=GFX1250-DIST %s
 
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
@@ -14,7 +14,7 @@
 ; Use .space to make the estimated MachineFunction size and final assembled
 ; size large without spelling out thousands of instructions.
 
-; A function below the 32 KiB kernel-descriptor prefetch limit must not use
+; A function below the 32 KiB explicit-prefetch threshold must not use
 ; explicit prefetch instructions.
 ; GFX1250-OBJ-LABEL: <below_threshold>:
 ; GFX1250-OBJ-NOT:   s_prefetch_inst_pc_rel
@@ -40,26 +40,55 @@ define amdgpu_kernel void @below_threshold() {
 ; NO-PREFETCH-NEXT:    .space 30000
 ; NO-PREFETCH-NEXT:    ;;#ASMEND
 ; NO-PREFETCH-NEXT:    s_endpgm
+;
+; GFX1250-DIST-LABEL: below_threshold:
+; GFX1250-DIST:       ; %bb.0:
+; GFX1250-DIST-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-DIST-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-DIST-NEXT:    v_nop
+; GFX1250-DIST-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-DIST-NEXT:    ;;#ASMSTART
+; GFX1250-DIST-NEXT:    .space 30000
+; GFX1250-DIST-NEXT:    ;;#ASMEND
+; GFX1250-DIST-NEXT:    s_endpgm
   call void asm sideeffect ".space 30000", ""()
   ret void
 }
 
-; The descriptor prefetches the first 32KiB; explicit requests cover the
+; The descriptor prefetches the first 8KiB; explicit requests cover the
 ; remaining code in the 64KiB I-cache window.
 ; GFX1250-OBJ-LABEL:      <partial_final_slot>:
-; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x7ff8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8ff0, null, 25
+; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x1ff8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2ff0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3fe8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4fe0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5fd8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6fd0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7fc8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8fc0, null, 25
 ; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x0, null, 0
 define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; GFX1250-LABEL: partial_final_slot:
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
 ; GFX1250-NEXT:  .Lpref_inst_offset0:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(64, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(64, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
 ; GFX1250-NEXT:  .Lpref_inst_offset1:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(96, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot), null, prefetchcachelines(96, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot)
 ; GFX1250-NEXT:  .Lpref_inst_offset2:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot), null, prefetchcachelines(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset3:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset3-partial_final_slot), null, prefetchcachelines(160, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset3-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset4:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset4-partial_final_slot), null, prefetchcachelines(192, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset4-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset5:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot), null, prefetchcachelines(224, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset6:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot), null, prefetchcachelines(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset7:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset7-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset7-partial_final_slot)
+; GFX1250-NEXT:  .Lpref_inst_offset8:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset8-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset8-partial_final_slot)
 ; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
@@ -87,42 +116,90 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; NO-PREFETCH-NEXT:    s_wait_kmcnt 0x0
 ; NO-PREFETCH-NEXT:    global_store_b32 v0, v0, s[0:1]
 ; NO-PREFETCH-NEXT:    s_endpgm
+;
+; GFX1250-DIST-LABEL: partial_final_slot:
+; GFX1250-DIST:       ; %bb.0:
+; GFX1250-DIST-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset0:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset1:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot), null, prefetchcachelines(160, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset2:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot), null, prefetchcachelines(192, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset3:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset3-partial_final_slot), null, prefetchcachelines(224, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset3-partial_final_slot)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset4:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset4-partial_final_slot), null, prefetchcachelines(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset4-partial_final_slot)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset5:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset6:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot)
+; GFX1250-DIST-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-DIST-NEXT:    v_nop
+; GFX1250-DIST-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-DIST-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
+; GFX1250-DIST-NEXT:    v_mov_b32_e32 v0, 0
+; GFX1250-DIST-NEXT:    ;;#ASMSTART
+; GFX1250-DIST-NEXT:    .space 40000
+; GFX1250-DIST-NEXT:    ;;#ASMEND
+; GFX1250-DIST-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-DIST-NEXT:    global_store_b32 v0, v0, s[0:1]
+; GFX1250-DIST-NEXT:    s_endpgm
+; GFX1250-DIST-NEXT:  .Lpref_func_end0:
   call void asm sideeffect ".space 40000", ""()
   store i32 0, ptr addrspace(1) %out
   ret void
 }
 
-; The 32KiB descriptor prefix leaves eight explicit requests for the complete
+; The 8KiB descriptor prefix leaves fourteen explicit requests for the complete
 ; 64KiB I-cache range.
 ; GFX1250-OBJ-LABEL:      <cache_size_limit>:
-; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x7ff8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8ff0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9fe8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xafe0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xbfd8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xcfd0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xdfc8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xefc0, null, 31
+; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x1ff8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2ff0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3fe8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4fe0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5fd8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6fd0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7fc8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8fc0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9fb8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xafb0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xbfa8, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xcfa0, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xdf98, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xef90, null, 31
 define amdgpu_kernel void @cache_size_limit() {
 ; GFX1250-LABEL: cache_size_limit:
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-NEXT:  .Lpref_inst_offset3:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset3-cache_size_limit), null, prefetchcachelines(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset3-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset4:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset4-cache_size_limit), null, prefetchcachelines(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset4-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset5:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset5-cache_size_limit), null, prefetchcachelines(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset5-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset6:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset6-cache_size_limit), null, prefetchcachelines(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset6-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset7:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit), null, prefetchcachelines(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit)
-; GFX1250-NEXT:  .Lpref_inst_offset8:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit), null, prefetchcachelines(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit)
 ; GFX1250-NEXT:  .Lpref_inst_offset9:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(64, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit), null, prefetchcachelines(64, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit)
 ; GFX1250-NEXT:  .Lpref_inst_offset10:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit)
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(96, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit), null, prefetchcachelines(96, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset11:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset11-cache_size_limit), null, prefetchcachelines(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset11-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset12:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset12-cache_size_limit), null, prefetchcachelines(160, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset12-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset13:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset13-cache_size_limit), null, prefetchcachelines(192, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset13-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset14:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset14-cache_size_limit), null, prefetchcachelines(224, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset14-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset15:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset15-cache_size_limit), null, prefetchcachelines(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset15-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset16:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset16-cache_size_limit), null, prefetchcachelines(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset16-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset17:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit), null, prefetchcachelines(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset18:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit), null, prefetchcachelines(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset19:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset19-cache_size_limit), null, prefetchcachelines(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset19-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset20:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset20-cache_size_limit), null, prefetchcachelines(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset20-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset21:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset21-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset21-cache_size_limit)
+; GFX1250-NEXT:  .Lpref_inst_offset22:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset22-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset22-cache_size_limit)
 ; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
@@ -142,26 +219,58 @@ define amdgpu_kernel void @cache_size_limit() {
 ; NO-PREFETCH-NEXT:    .space 65536
 ; NO-PREFETCH-NEXT:    ;;#ASMEND
 ; NO-PREFETCH-NEXT:    s_endpgm
+;
+; GFX1250-DIST-LABEL: cache_size_limit:
+; GFX1250-DIST:       ; %bb.0:
+; GFX1250-DIST-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset7:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit), null, prefetchcachelines(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset8:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit), null, prefetchcachelines(160, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset9:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit), null, prefetchcachelines(192, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset10:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit), null, prefetchcachelines(224, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset11:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset11-cache_size_limit), null, prefetchcachelines(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset11-cache_size_limit)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset12:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset12-cache_size_limit), null, prefetchcachelines(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset12-cache_size_limit)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset13:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset13-cache_size_limit), null, prefetchcachelines(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset13-cache_size_limit)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset14:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset14-cache_size_limit), null, prefetchcachelines(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset14-cache_size_limit)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset15:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset15-cache_size_limit), null, prefetchcachelines(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset15-cache_size_limit)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset16:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset16-cache_size_limit), null, prefetchcachelines(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset16-cache_size_limit)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset17:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset18:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit)
+; GFX1250-DIST-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-DIST-NEXT:    v_nop
+; GFX1250-DIST-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-DIST-NEXT:    ;;#ASMSTART
+; GFX1250-DIST-NEXT:    .space 65536
+; GFX1250-DIST-NEXT:    ;;#ASMEND
+; GFX1250-DIST-NEXT:    s_endpgm
+; GFX1250-DIST-NEXT:  .Lpref_func_end1:
   call void asm sideeffect ".space 65536", ""()
   ret void
 }
 
-; With the default 32KiB descriptor prefix, explicit prefetches are deferred
-; to the shared exit block.
+; With a 16KiB descriptor prefix override, explicit prefetches are deferred to
+; the shared exit block.
 declare i32 @llvm.amdgcn.workgroup.id.x()
 
 define amdgpu_kernel void @postdominated_prefetch() {
-; GFX1250-DIST-LABEL: postdominated_prefetch:
-; GFX1250-DIST-NOT:   s_prefetch_inst_pc_rel
-; GFX1250-DIST:       .LBB3_3: ; %Flow
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset
-; GFX1250-DIST-NEXT:  s_prefetch_inst_pc_rel prefetchoffset(128,
-; GFX1250-DIST:       .LBB3_5: ; %join
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset
-; GFX1250-DIST-NEXT:  s_prefetch_inst_pc_rel prefetchoffset(192,
 ; GFX1250-LABEL: postdominated_prefetch:
 ; GFX1250:       ; %bb.0: ; %entry
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:  .Lpref_inst_offset23:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(64, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch), null, prefetchcachelines(64, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset24:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(96, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch), null, prefetchcachelines(96, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch)
 ; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
@@ -185,6 +294,10 @@ define amdgpu_kernel void @postdominated_prefetch() {
 ; GFX1250-NEXT:  .LBB3_2:
 ; GFX1250-NEXT:    s_mov_b32 s0, -1
 ; GFX1250-NEXT:  .LBB3_3: ; %Flow
+; GFX1250-NEXT:  .Lpref_inst_offset25:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset25-postdominated_prefetch), null, prefetchcachelines(128, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset25-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset26:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset26-postdominated_prefetch), null, prefetchcachelines(160, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset26-postdominated_prefetch)
 ; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
 ; GFX1250-NEXT:    s_and_b32 s0, s0, exec_lo
 ; GFX1250-NEXT:    s_cselect_b32 s0, 1, 0
@@ -195,16 +308,20 @@ define amdgpu_kernel void @postdominated_prefetch() {
 ; GFX1250-NEXT:    .space 8000
 ; GFX1250-NEXT:    ;;#ASMEND
 ; GFX1250-NEXT:  .LBB3_5: ; %join
-; GFX1250-NEXT:  .Lpref_inst_offset11:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset11-postdominated_prefetch), null, prefetchcachelines(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset11-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset12:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset12-postdominated_prefetch), null, prefetchcachelines(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset12-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset13:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset13-postdominated_prefetch), null, prefetchcachelines(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset13-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset14:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset14-postdominated_prefetch), null, prefetchcachelines(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset14-postdominated_prefetch)
-; GFX1250-NEXT:  .Lpref_inst_offset15:
-; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset15-postdominated_prefetch), null, prefetchcachelines(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset15-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset27:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset27-postdominated_prefetch), null, prefetchcachelines(192, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset27-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset28:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset28-postdominated_prefetch), null, prefetchcachelines(224, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset28-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset29:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset29-postdominated_prefetch), null, prefetchcachelines(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset29-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset30:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset30-postdominated_prefetch), null, prefetchcachelines(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset30-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset31:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset31-postdominated_prefetch), null, prefetchcachelines(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset31-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset32:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset32-postdominated_prefetch), null, prefetchcachelines(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset32-postdominated_prefetch)
+; GFX1250-NEXT:  .Lpref_inst_offset33:
+; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset33-postdominated_prefetch), null, prefetchcachelines(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset33-postdominated_prefetch)
 ; GFX1250-NEXT:    ;;#ASMSTART
 ; GFX1250-NEXT:    .space 32000
 ; GFX1250-NEXT:    ;;#ASMEND
@@ -251,6 +368,66 @@ define amdgpu_kernel void @postdominated_prefetch() {
 ; NO-PREFETCH-NEXT:    .space 32000
 ; NO-PREFETCH-NEXT:    ;;#ASMEND
 ; NO-PREFETCH-NEXT:    s_endpgm
+;
+; GFX1250-DIST-LABEL: postdominated_prefetch:
+; GFX1250-DIST:       ; %bb.0: ; %entry
+; GFX1250-DIST-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-DIST-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-DIST-NEXT:    v_nop
+; GFX1250-DIST-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-DIST-NEXT:    s_bfe_u32 s0, ttmp6, 0x4000c
+; GFX1250-DIST-NEXT:    s_and_b32 s1, ttmp6, 15
+; GFX1250-DIST-NEXT:    s_add_co_i32 s0, s0, 1
+; GFX1250-DIST-NEXT:    s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
+; GFX1250-DIST-NEXT:    s_mul_i32 s0, ttmp9, s0
+; GFX1250-DIST-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-DIST-NEXT:    s_add_co_i32 s1, s1, s0
+; GFX1250-DIST-NEXT:    s_cmp_eq_u32 s2, 0
+; GFX1250-DIST-NEXT:    s_cselect_b32 s0, ttmp9, s1
+; GFX1250-DIST-NEXT:    s_cmp_lg_u32 s0, 0
+; GFX1250-DIST-NEXT:    s_mov_b32 s0, 0
+; GFX1250-DIST-NEXT:    s_cbranch_scc0 .LBB3_2
+; GFX1250-DIST-NEXT:  ; %bb.1: ; %else
+; GFX1250-DIST-NEXT:    ;;#ASMSTART
+; GFX1250-DIST-NEXT:    .space 8000
+; GFX1250-DIST-NEXT:    ;;#ASMEND
+; GFX1250-DIST-NEXT:    s_branch .LBB3_3
+; GFX1250-DIST-NEXT:  .LBB3_2:
+; GFX1250-DIST-NEXT:    s_mov_b32 s0, -1
+; GFX1250-DIST-NEXT:  .LBB3_3: ; %Flow
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset19:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset19-postdominated_prefetch), null, prefetchcachelines(128, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset19-postdominated_prefetch)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset20:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset20-postdominated_prefetch), null, prefetchcachelines(160, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset20-postdominated_prefetch)
+; GFX1250-DIST-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-DIST-NEXT:    s_and_b32 s0, s0, exec_lo
+; GFX1250-DIST-NEXT:    s_cselect_b32 s0, 1, 0
+; GFX1250-DIST-NEXT:    s_cmp_lg_u32 s0, 1
+; GFX1250-DIST-NEXT:    s_cbranch_scc1 .LBB3_5
+; GFX1250-DIST-NEXT:  ; %bb.4: ; %then
+; GFX1250-DIST-NEXT:    ;;#ASMSTART
+; GFX1250-DIST-NEXT:    .space 8000
+; GFX1250-DIST-NEXT:    ;;#ASMEND
+; GFX1250-DIST-NEXT:  .LBB3_5: ; %join
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset21:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset21-postdominated_prefetch), null, prefetchcachelines(192, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset21-postdominated_prefetch)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset22:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset22-postdominated_prefetch), null, prefetchcachelines(224, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset22-postdominated_prefetch)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset23:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch), null, prefetchcachelines(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset24:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch), null, prefetchcachelines(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset25:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset25-postdominated_prefetch), null, prefetchcachelines(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset25-postdominated_prefetch)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset26:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset26-postdominated_prefetch), null, prefetchcachelines(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset26-postdominated_prefetch)
+; GFX1250-DIST-NEXT:  .Lpref_inst_offset27:
+; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset27-postdominated_prefetch), null, prefetchcachelines(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset27-postdominated_prefetch)
+; GFX1250-DIST-NEXT:    ;;#ASMSTART
+; GFX1250-DIST-NEXT:    .space 32000
+; GFX1250-DIST-NEXT:    ;;#ASMEND
+; GFX1250-DIST-NEXT:    s_endpgm
+; GFX1250-DIST-NEXT:  .Lpref_func_end2:
 entry:
   %id = call i32 @llvm.amdgcn.workgroup.id.x()
   %cond = icmp eq i32 %id, 0
@@ -289,6 +466,15 @@ define void @called_function() {
 ; NO-PREFETCH-NEXT:    .space 40000
 ; NO-PREFETCH-NEXT:    ;;#ASMEND
 ; NO-PREFETCH-NEXT:    s_set_pc_i64 s[30:31]
+;
+; GFX1250-DIST-LABEL: called_function:
+; GFX1250-DIST:       ; %bb.0:
+; GFX1250-DIST-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-DIST-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-DIST-NEXT:    ;;#ASMSTART
+; GFX1250-DIST-NEXT:    .space 40000
+; GFX1250-DIST-NEXT:    ;;#ASMEND
+; GFX1250-DIST-NEXT:    s_set_pc_i64 s[30:31]
   call void asm sideeffect ".space 40000", ""()
   ret void
 }

>From 0d9a6b8d8c03837a87c1bcbfab1f5672f466c4a6 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Wed, 30 Sep 2026 09:02:42 -0500
Subject: [PATCH 17/22] Update BB prologue matcher

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 .../AMDGPU/AMDGPUInsertICachePrefetch.cpp     | 12 +--
 .../CodeGen/AMDGPU/icache-prefetch-config.ll  | 10 ++-
 llvm/test/CodeGen/AMDGPU/icache-prefetch.ll   | 74 +++++++++----------
 3 files changed, 52 insertions(+), 44 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index 73766a35e2ed2..be39954a6dec8 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -159,11 +159,13 @@ static MachineBasicBlock::iterator findMBBInsertionPoint(MachineBasicBlock &MBB,
       continue;
     }
     if (!SkippedInitialUnclausedVmemPrologue &&
-        InsertPt->getOpcode() == AMDGPU::GLOBAL_PREFETCH_B8_SADDR) {
-      auto Next = InsertPt;
-      ++Next;
-      if (Next != MBB.end() && Next->getOpcode() == AMDGPU::V_NOP_e32) {
-        InsertPt = ++Next;
+        InsertPt->getOpcode() == AMDGPU::S_MOV_B64) {
+      auto Vnop = std::next(InsertPt);
+      auto GlobalPrefetch = Vnop == MBB.end() ? MBB.end() : std::next(Vnop);
+      if (Vnop != MBB.end() && Vnop->getOpcode() == AMDGPU::V_NOP_e32 &&
+          GlobalPrefetch != MBB.end() &&
+          GlobalPrefetch->getOpcode() == AMDGPU::GLOBAL_PREFETCH_B8_SADDR) {
+        InsertPt = std::next(GlobalPrefetch);
         SkippedInitialUnclausedVmemPrologue = true;
         continue;
       }
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
index 574ca53a26363..edb853f4f2d64 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
@@ -68,11 +68,17 @@ define amdgpu_kernel void @between_override_thresholds() {
 }
 
 ; INITIAL-16K-LABEL: cache_size_limit:
+; INITIAL-16K:          s_mov_b64
+; INITIAL-16K-NEXT:     v_nop
+; INITIAL-16K-NEXT:     global_prefetch_b8
 ; INITIAL-16K-COUNT-12: s_prefetch_inst_pc_rel
-; INITIAL-16K-NEXT:     s_mov_b64
+; INITIAL-16K-NEXT:     ;;#ASMSTART
 ; GFX1250-LABEL:     cache_size_limit:
+; GFX1250:           s_mov_b64
+; GFX1250-NEXT:      v_nop
+; GFX1250-NEXT:      global_prefetch_b8
 ; GFX1250-COUNT-14:  s_prefetch_inst_pc_rel
-; GFX1250-NEXT:      s_mov_b64
+; GFX1250-NEXT:      ;;#ASMSTART
 define amdgpu_kernel void @cache_size_limit() {
   call void asm sideeffect ".space 65536", ""()
   ret void
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 286cfb8cb5949..628579dcbcacb 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -58,19 +58,22 @@ define amdgpu_kernel void @below_threshold() {
 ; The descriptor prefetches the first 8KiB; explicit requests cover the
 ; remaining code in the 64KiB I-cache window.
 ; GFX1250-OBJ-LABEL:      <partial_final_slot>:
-; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x1ff8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2ff0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3fe8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4fe0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5fd8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6fd0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7fc8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8fc0, null, 25
+; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x1fe4, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2fdc, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3fd4, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4fcc, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5fc4, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6fbc, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7fb4, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8fac, null, 25
 ; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x0, null, 0
 define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; GFX1250-LABEL: partial_final_slot:
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:  .Lpref_inst_offset0:
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(64, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(64, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
 ; GFX1250-NEXT:  .Lpref_inst_offset1:
@@ -89,9 +92,6 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset7-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset7-partial_final_slot)
 ; GFX1250-NEXT:  .Lpref_inst_offset8:
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset8-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset8-partial_final_slot)
-; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
-; GFX1250-NEXT:    v_nop
-; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
 ; GFX1250-NEXT:    v_mov_b32_e32 v0, 0
 ; GFX1250-NEXT:    ;;#ASMSTART
@@ -120,6 +120,9 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; GFX1250-DIST-LABEL: partial_final_slot:
 ; GFX1250-DIST:       ; %bb.0:
 ; GFX1250-DIST-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-DIST-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-DIST-NEXT:    v_nop
+; GFX1250-DIST-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-DIST-NEXT:  .Lpref_inst_offset0:
 ; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
 ; GFX1250-DIST-NEXT:  .Lpref_inst_offset1:
@@ -134,9 +137,6 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot)
 ; GFX1250-DIST-NEXT:  .Lpref_inst_offset6:
 ; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot)
-; GFX1250-DIST-NEXT:    s_mov_b64 s[64:65], 0
-; GFX1250-DIST-NEXT:    v_nop
-; GFX1250-DIST-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-DIST-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
 ; GFX1250-DIST-NEXT:    v_mov_b32_e32 v0, 0
 ; GFX1250-DIST-NEXT:    ;;#ASMSTART
@@ -154,24 +154,27 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; The 8KiB descriptor prefix leaves fourteen explicit requests for the complete
 ; 64KiB I-cache range.
 ; GFX1250-OBJ-LABEL:      <cache_size_limit>:
-; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x1ff8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2ff0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3fe8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4fe0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5fd8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6fd0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7fc8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8fc0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9fb8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xafb0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xbfa8, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xcfa0, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xdf98, null, 31
-; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xef90, null, 31
+; GFX1250-OBJ:            s_prefetch_inst_pc_rel 0x1fe4, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x2fdc, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x3fd4, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x4fcc, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x5fc4, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x6fbc, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x7fb4, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x8fac, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0x9fa4, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xaf9c, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xbf94, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xcf8c, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xdf84, null, 31
+; GFX1250-OBJ-NEXT:       s_prefetch_inst_pc_rel 0xef7c, null, 31
 define amdgpu_kernel void @cache_size_limit() {
 ; GFX1250-LABEL: cache_size_limit:
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:  .Lpref_inst_offset9:
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(64, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit), null, prefetchcachelines(64, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit)
 ; GFX1250-NEXT:  .Lpref_inst_offset10:
@@ -200,9 +203,6 @@ define amdgpu_kernel void @cache_size_limit() {
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset21-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset21-cache_size_limit)
 ; GFX1250-NEXT:  .Lpref_inst_offset22:
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset22-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset22-cache_size_limit)
-; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
-; GFX1250-NEXT:    v_nop
-; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    ;;#ASMSTART
 ; GFX1250-NEXT:    .space 65536
 ; GFX1250-NEXT:    ;;#ASMEND
@@ -223,6 +223,9 @@ define amdgpu_kernel void @cache_size_limit() {
 ; GFX1250-DIST-LABEL: cache_size_limit:
 ; GFX1250-DIST:       ; %bb.0:
 ; GFX1250-DIST-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-DIST-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-DIST-NEXT:    v_nop
+; GFX1250-DIST-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-DIST-NEXT:  .Lpref_inst_offset7:
 ; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit), null, prefetchcachelines(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit)
 ; GFX1250-DIST-NEXT:  .Lpref_inst_offset8:
@@ -247,9 +250,6 @@ define amdgpu_kernel void @cache_size_limit() {
 ; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit)
 ; GFX1250-DIST-NEXT:  .Lpref_inst_offset18:
 ; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit)
-; GFX1250-DIST-NEXT:    s_mov_b64 s[64:65], 0
-; GFX1250-DIST-NEXT:    v_nop
-; GFX1250-DIST-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-DIST-NEXT:    ;;#ASMSTART
 ; GFX1250-DIST-NEXT:    .space 65536
 ; GFX1250-DIST-NEXT:    ;;#ASMEND
@@ -267,13 +267,13 @@ define amdgpu_kernel void @postdominated_prefetch() {
 ; GFX1250-LABEL: postdominated_prefetch:
 ; GFX1250:       ; %bb.0: ; %entry
 ; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:  .Lpref_inst_offset23:
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(64, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch), null, prefetchcachelines(64, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch)
 ; GFX1250-NEXT:  .Lpref_inst_offset24:
 ; GFX1250-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(96, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch), null, prefetchcachelines(96, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch)
-; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
-; GFX1250-NEXT:    v_nop
-; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_bfe_u32 s0, ttmp6, 0x4000c
 ; GFX1250-NEXT:    s_and_b32 s1, ttmp6, 15
 ; GFX1250-NEXT:    s_add_co_i32 s0, s0, 1

>From 634bab5e18ee98e4f639745947cfba070baf78ba Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 1 Oct 2026 02:06:13 -0500
Subject: [PATCH 18/22] Roundtrip ICachePrefetchLines through YAML

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp              | 2 ++
 llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h                | 2 ++
 llvm/test/CodeGen/AMDGPU/icache-prefetch.ll                   | 4 ++++
 llvm/test/CodeGen/MIR/AMDGPU/long-branch-reg-all-sgpr-used.ll | 2 ++
 .../CodeGen/MIR/AMDGPU/machine-function-info-after-pei.ll     | 1 +
 .../MIR/AMDGPU/machine-function-info-long-branch-reg-debug.ll | 1 +
 .../MIR/AMDGPU/machine-function-info-long-branch-reg.ll       | 1 +
 llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-no-ir.mir  | 4 ++++
 llvm/test/CodeGen/MIR/AMDGPU/machine-function-info.ll         | 4 ++++
 9 files changed, 21 insertions(+)

diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp
index d1df50d26a832..ffbf1151dae48 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp
@@ -750,6 +750,7 @@ yaml::SIMachineFunctionInfo::SIMachineFunctionInfo(
       DynamicVGPRBlockSize(MFI.getDynamicVGPRBlockSize()),
       ScratchReservedForDynamicVGPRs(MFI.getScratchReservedForDynamicVGPRs()),
       NumKernargPreloadSGPRs(MFI.getNumKernargPreloadedSGPRs()),
+      ICachePrefetchLines(MFI.getICachePrefetchLines()),
       MinNumAGPRs(MFI.getMinNumAGPRs()) {
   for (Register Reg : MFI.getSGPRSpillPhysVGPRs())
     SpillPhysVGPRS.push_back(regToString(Reg, TRI));
@@ -797,6 +798,7 @@ bool SIMachineFunctionInfo::initializeBaseYamlFields(
   NumWaveDispatchVGPRs = YamlMFI.NumWaveDispatchVGPRs;
   BytesInStackArgArea = YamlMFI.BytesInStackArgArea;
   ReturnsVoid = YamlMFI.ReturnsVoid;
+  ICachePrefetchLines = YamlMFI.ICachePrefetchLines;
   IsWholeWaveFunction = YamlMFI.IsWholeWaveFunction;
   MinNumAGPRs = YamlMFI.MinNumAGPRs;
   // This can also be set by the function attribute, MFI has higher precedence
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
index 78240ecd67e37..6e58ac3283318 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
@@ -309,6 +309,7 @@ struct SIMachineFunctionInfo final : public yaml::MachineFunctionInfo {
   unsigned ScratchReservedForDynamicVGPRs = 0;
 
   unsigned NumKernargPreloadSGPRs = 0;
+  unsigned ICachePrefetchLines = 0;
 
   unsigned MinNumAGPRs = ~0u;
 
@@ -370,6 +371,7 @@ template <> struct MappingTraits<SIMachineFunctionInfo> {
     YamlIO.mapOptional("scratchReservedForDynamicVGPRs",
                        MFI.ScratchReservedForDynamicVGPRs, 0);
     YamlIO.mapOptional("numKernargPreloadSGPRs", MFI.NumKernargPreloadSGPRs, 0);
+    YamlIO.mapOptional("iCachePrefetchLines", MFI.ICachePrefetchLines, 0u);
     YamlIO.mapOptional("isWholeWaveFunction", MFI.IsWholeWaveFunction, false);
     YamlIO.mapOptional("minNumAGPRs", MFI.MinNumAGPRs, ~0u);
   }
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 628579dcbcacb..796702bda1c53 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -11,6 +11,10 @@
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
 ; RUN: llvm-objdump -d %t.o | FileCheck -check-prefix=GFX1250-OBJ %s
 
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -stop-after=amdgpu-insert-icache-prefetch -o %t.mir %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -start-after=amdgpu-insert-icache-prefetch -filetype=obj -o %t-resumed.o %t.mir
+; RUN: llvm-objdump -d %t-resumed.o | FileCheck -check-prefix=GFX1250-OBJ %s
+
 ; Use .space to make the estimated MachineFunction size and final assembled
 ; size large without spelling out thousands of instructions.
 
diff --git a/llvm/test/CodeGen/MIR/AMDGPU/long-branch-reg-all-sgpr-used.ll b/llvm/test/CodeGen/MIR/AMDGPU/long-branch-reg-all-sgpr-used.ll
index 025008fac5144..752e353fd440c 100644
--- a/llvm/test/CodeGen/MIR/AMDGPU/long-branch-reg-all-sgpr-used.ll
+++ b/llvm/test/CodeGen/MIR/AMDGPU/long-branch-reg-all-sgpr-used.ll
@@ -49,6 +49,7 @@
 ; CHECK-NEXT:   dynamicVGPRBlockSize: 0
 ; CHECK-NEXT:   scratchReservedForDynamicVGPRs: 0
 ; CHECK-NEXT:   numKernargPreloadSGPRs: 0
+; CHECK-NEXT:   iCachePrefetchLines: 0
 ; CHECK-NEXT:   isWholeWaveFunction: false
 ; CHECK-NEXT:   minNumAGPRs: 4294967295
 ; CHECK-NEXT: body:
@@ -323,6 +324,7 @@
 ; CHECK-NEXT:   dynamicVGPRBlockSize: 0
 ; CHECK-NEXT:   scratchReservedForDynamicVGPRs: 0
 ; CHECK-NEXT:   numKernargPreloadSGPRs: 0
+; CHECK-NEXT:   iCachePrefetchLines: 0
 ; CHECK-NEXT:   isWholeWaveFunction: false
 ; CHECK-NEXT:   minNumAGPRs: 4294967295
 ; CHECK-NEXT: body:
diff --git a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-after-pei.ll b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-after-pei.ll
index 2c31f5c9c3477..6078926115f06 100644
--- a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-after-pei.ll
+++ b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-after-pei.ll
@@ -49,6 +49,7 @@
 ; AFTER-PEI-NEXT: dynamicVGPRBlockSize: 0
 ; AFTER-PEI-NEXT: scratchReservedForDynamicVGPRs: 0
 ; AFTER-PEI-NEXT: numKernargPreloadSGPRs: 0
+; AFTER-PEI-NEXT: iCachePrefetchLines: 0
 ; AFTER-PEI-NEXT: isWholeWaveFunction: false
 ; AFTER-PEI-NEXT: minNumAGPRs: 4294967295
 ; AFTER-PEI-NEXT: body:
diff --git a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg-debug.ll b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg-debug.ll
index 5497316fc1936..32e1a7ebb9a2d 100644
--- a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg-debug.ll
+++ b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg-debug.ll
@@ -49,6 +49,7 @@
 ; CHECK-NEXT: dynamicVGPRBlockSize: 0
 ; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
 ; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
 ; CHECK-NEXT: isWholeWaveFunction: false
 ; CHECK-NEXT: minNumAGPRs: 4294967295
 ; CHECK-NEXT: body:
diff --git a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg.ll b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg.ll
index 8aab1c0fa55c8..23e3ea0425c4b 100644
--- a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg.ll
+++ b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-long-branch-reg.ll
@@ -49,6 +49,7 @@
 ; CHECK-NEXT: dynamicVGPRBlockSize: 0
 ; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
 ; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
 ; CHECK-NEXT: isWholeWaveFunction: false
 ; CHECK-NEXT: minNumAGPRs: 4294967295
 ; CHECK-NEXT: body:
diff --git a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-no-ir.mir b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-no-ir.mir
index 9d1236b0f3739..a6e33ca9779e5 100644
--- a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-no-ir.mir
+++ b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info-no-ir.mir
@@ -58,6 +58,7 @@
 # FULL-NEXT: dynamicVGPRBlockSize: 0
 # FULL-NEXT: scratchReservedForDynamicVGPRs: 0
 # FULL-NEXT: numKernargPreloadSGPRs: 0
+# FULL-NEXT: iCachePrefetchLines: 0
 # FULL-NEXT: isWholeWaveFunction: false
 # FULL-NEXT: minNumAGPRs: 4294967295
 # FULL-NEXT: body:
@@ -170,6 +171,7 @@ body:             |
 # FULL-NEXT: dynamicVGPRBlockSize: 0
 # FULL-NEXT: scratchReservedForDynamicVGPRs: 0
 # FULL-NEXT: numKernargPreloadSGPRs: 0
+# FULL-NEXT: iCachePrefetchLines: 0
 # FULL-NEXT: isWholeWaveFunction: false
 # FULL-NEXT: minNumAGPRs: 4294967295
 # FULL-NEXT: body:
@@ -254,6 +256,7 @@ body:             |
 # FULL-NEXT: dynamicVGPRBlockSize: 0
 # FULL-NEXT: scratchReservedForDynamicVGPRs: 0
 # FULL-NEXT: numKernargPreloadSGPRs: 0
+# FULL-NEXT: iCachePrefetchLines: 0
 # FULL-NEXT: isWholeWaveFunction: false
 # FULL-NEXT: minNumAGPRs: 4294967295
 # FULL-NEXT: body:
@@ -339,6 +342,7 @@ body:             |
 # FULL-NEXT: dynamicVGPRBlockSize: 0
 # FULL-NEXT: scratchReservedForDynamicVGPRs: 0
 # FULL-NEXT: numKernargPreloadSGPRs: 0
+# FULL-NEXT: iCachePrefetchLines: 0
 # FULL-NEXT: isWholeWaveFunction: false
 # FULL-NEXT: minNumAGPRs: 4294967295
 # FULL-NEXT: body:
diff --git a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info.ll b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info.ll
index 770718c166d4c..d38139d11a6f2 100644
--- a/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info.ll
+++ b/llvm/test/CodeGen/MIR/AMDGPU/machine-function-info.ll
@@ -59,6 +59,7 @@
 ; CHECK-NEXT: dynamicVGPRBlockSize: 0
 ; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
 ; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
 ; CHECK-NEXT: isWholeWaveFunction: false
 ; CHECK-NEXT: minNumAGPRs: 4294967295
 ; CHECK-NEXT: body:
@@ -113,6 +114,7 @@ define amdgpu_kernel void @kernel(i32 %arg0, i64 %arg1, <16 x i32> %arg2) {
 ; CHECK-NEXT: dynamicVGPRBlockSize: 0
 ; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
 ; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
 ; CHECK-NEXT: isWholeWaveFunction: false
 ; CHECK-NEXT: minNumAGPRs: 4294967295
 ; CHECK-NEXT: body:
@@ -191,6 +193,7 @@ define amdgpu_ps void @gds_size_shader(i32 %arg0, i32 inreg %arg1) #5 {
 ; CHECK-NEXT: dynamicVGPRBlockSize: 0
 ; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
 ; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
 ; CHECK-NEXT: isWholeWaveFunction: false
 ; CHECK-NEXT: minNumAGPRs: 4294967295
 ; CHECK-NEXT: body:
@@ -251,6 +254,7 @@ define void @function() {
 ; CHECK-NEXT: dynamicVGPRBlockSize: 0
 ; CHECK-NEXT: scratchReservedForDynamicVGPRs: 0
 ; CHECK-NEXT: numKernargPreloadSGPRs: 0
+; CHECK-NEXT: iCachePrefetchLines: 0
 ; CHECK-NEXT: isWholeWaveFunction: false
 ; CHECK-NEXT: minNumAGPRs: 4294967295
 ; CHECK-NEXT: body:

>From e4216b4b5b8e96c48f5004279b3b5621771aa168 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 1 Oct 2026 02:45:27 -0500
Subject: [PATCH 19/22] Fix off-by-one for INST_PREF_SIZE

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp   |  3 +--
 .../AMDGPU/AMDGPUInsertICachePrefetch.cpp     |  3 ++-
 .../CodeGen/AMDGPU/icache-prefetch-config.ll  | 27 +++++++++++--------
 3 files changed, 19 insertions(+), 14 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 3776390d8deb2..e765cd0c76993 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -273,8 +273,7 @@ void AMDGPUAsmPrinter::endFunction(const MachineFunction *MF) {
 
     const MCExpr *InstPrefSize;
     if (MFI.hasICachePrefetch()) {
-      InstPrefSize =
-          MCConstantExpr::create(MFI.getICachePrefetchLines() - 1, Ctx);
+      InstPrefSize = MCConstantExpr::create(MFI.getICachePrefetchLines(), Ctx);
     } else {
       const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
           MCSymbolRefExpr::create(getFunctionEnd(), OutContext),
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
index be39954a6dec8..196409edc872e 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertICachePrefetch.cpp
@@ -104,7 +104,8 @@ static ICachePrefetchConfig getICachePrefetchConfig(const GCNSubtarget &ST) {
 
   uint32_t Mask, Shift, Width, CacheLineSize;
   ST.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
-  uint64_t DescriptorPrefetchCapacity = (uint64_t{1} << Width) * CacheLineSize;
+  uint64_t DescriptorPrefetchCapacity =
+      ((uint64_t{1} << Width) - 1) * CacheLineSize;
   uint64_t InitialSize = ICachePrefetchInitialSize.getNumOccurrences()
                              ? ICachePrefetchInitialSize
                              : ST.getInitialInstPrefSize();
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
index edb853f4f2d64..3006a1ca588b3 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
@@ -1,31 +1,36 @@
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -o - %s | FileCheck -check-prefix=GFX1200 %s
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -o - %s | FileCheck -check-prefix=GFX1250 %s
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -o - %s | FileCheck -check-prefix=INITIAL-16K %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=32640 -o - %s | FileCheck -check-prefix=MAX-INITIAL %s
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=49152 -o - %s | FileCheck -check-prefix=THRESHOLD-48K %s
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=THRESHOLD-16K %s
 ; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=EQUAL-16K %s
 ; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=0 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
 ; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=8193 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=32896 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
+; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=32768 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
 ; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=0 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
 ; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=32769 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
 ; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=65664 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
 ; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=8192 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-ORDER %s
 
-; The default threshold is the 32KiB descriptor capacity on both targets.
+; The default threshold is the 32640-byte descriptor capacity on both targets.
 ; gfx1200 initially prefetches 16KiB, while gfx1250 initially prefetches 8KiB.
 define amdgpu_kernel void @above_default_threshold() {
 ; GFX1200-LABEL: above_default_threshold:
 ; GFX1200:       s_prefetch_inst_pc_rel prefetchoffset(128,
-; GFX1200:       .amdhsa_inst_pref_size 127
+; GFX1200:       .amdhsa_inst_pref_size 128
 ;
 ; GFX1250-LABEL: above_default_threshold:
 ; GFX1250:       s_prefetch_inst_pc_rel prefetchoffset(64,
-; GFX1250:       .amdhsa_inst_pref_size 63
+; GFX1250:       .amdhsa_inst_pref_size 64
 ;
 ; INITIAL-16K-LABEL: above_default_threshold:
 ; INITIAL-16K:       s_prefetch_inst_pc_rel prefetchoffset(128,
-; INITIAL-16K:       .amdhsa_inst_pref_size 127
+; INITIAL-16K:       .amdhsa_inst_pref_size 128
+;
+; MAX-INITIAL-LABEL: above_default_threshold:
+; MAX-INITIAL:       s_prefetch_inst_pc_rel prefetchoffset(255,
+; MAX-INITIAL:       .amdhsa_inst_pref_size 255
 ;
 ; THRESHOLD-48K-LABEL: above_default_threshold:
 ; THRESHOLD-48K-NOT:   s_prefetch_inst_pc_rel
@@ -36,19 +41,19 @@ define amdgpu_kernel void @above_default_threshold() {
 }
 
 ; The gfx1250 prologue and s_endpgm add 32 bytes, making these functions
-; exactly 32KiB and four bytes over 32KiB respectively.
+; exactly 32640 bytes and four bytes over 32640 bytes respectively.
 define amdgpu_kernel void @at_default_threshold() {
 ; GFX1250-LABEL: at_default_threshold:
 ; GFX1250-NOT:   s_prefetch_inst_pc_rel
 ; GFX1250:       s_endpgm
-  call void asm sideeffect ".space 32736", ""()
+  call void asm sideeffect ".space 32608", ""()
   ret void
 }
 
 define amdgpu_kernel void @above_default_threshold_by_four() {
 ; GFX1250-LABEL: above_default_threshold_by_four:
 ; GFX1250:       s_prefetch_inst_pc_rel prefetchoffset(64,
-  call void asm sideeffect ".space 32740", ""()
+  call void asm sideeffect ".space 32612", ""()
   ret void
 }
 
@@ -58,11 +63,11 @@ define amdgpu_kernel void @above_default_threshold_by_four() {
 define amdgpu_kernel void @between_override_thresholds() {
 ; THRESHOLD-16K-LABEL: between_override_thresholds:
 ; THRESHOLD-16K:       s_prefetch_inst_pc_rel prefetchoffset(64,
-; THRESHOLD-16K:       .amdhsa_inst_pref_size 63
+; THRESHOLD-16K:       .amdhsa_inst_pref_size 64
 ;
 ; EQUAL-16K-LABEL: between_override_thresholds:
 ; EQUAL-16K:       s_prefetch_inst_pc_rel prefetchoffset(128,
-; EQUAL-16K:       .amdhsa_inst_pref_size 127
+; EQUAL-16K:       .amdhsa_inst_pref_size 128
   call void asm sideeffect ".space 20000", ""()
   ret void
 }
@@ -84,6 +89,6 @@ define amdgpu_kernel void @cache_size_limit() {
   ret void
 }
 
-; INVALID-INITIAL: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must be a non-zero multiple of 128 bytes not exceeding 32768 bytes for gfx1250
+; INVALID-INITIAL: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must be a non-zero multiple of 128 bytes not exceeding 32640 bytes for gfx1250
 ; INVALID-THRESHOLD: LLVM ERROR: -amdgpu-icache-prefetch-threshold must be a non-zero multiple of 128 bytes not exceeding 65536 bytes for gfx1250
 ; INVALID-ORDER: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must not exceed -amdgpu-icache-prefetch-threshold for gfx1250

>From 8e6947c0d48c747a33dddf8def3771e6816a242a Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 1 Oct 2026 03:25:08 -0500
Subject: [PATCH 20/22] Remove outdated code

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h | 3 ---
 llvm/lib/Target/AMDGPU/SIDefines.h        | 3 ---
 2 files changed, 6 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
index 871bd738b947b..ca82af08ffade 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.h
@@ -165,9 +165,6 @@ class AMDGPUAsmPrinter final : public AsmPrinter {
   /// Get the symbol for the function end, used for ICache prefetch MCExprs.
   MCSymbol *getPrefetchEndSym() const { return PrefetchEndSym; }
 
-  /// Get the code size estimate from SIProgramInfo.
-  uint64_t getCodeSize() { return CurrentProgramInfo.getFunctionCodeSize(*MF); }
-
 protected:
   void getAnalysisUsage(AnalysisUsage &AU) const override;
 
diff --git a/llvm/lib/Target/AMDGPU/SIDefines.h b/llvm/lib/Target/AMDGPU/SIDefines.h
index a67c259541378..864a84cc8daf4 100644
--- a/llvm/lib/Target/AMDGPU/SIDefines.h
+++ b/llvm/lib/Target/AMDGPU/SIDefines.h
@@ -824,9 +824,6 @@ enum ModeRegisterMasks : uint32_t {
   SRC2_VGPR_MSB = 0x3 << 18,
   VGPR_MSB_MASK = 0xff << 12, // Bits 12..19
 
-  // GFX1250 instruction prefetch
-  SCALAR_PREFETCH_EN = 1 << 24,
-
   REPLAY_MODE = 1 << 25,
   FLAT_SCRATCH_IS_NV = 1 << 26,
 };

>From 83f108b2960f46c15b8dab0a9cea279be3dd8f5c Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 1 Oct 2026 03:25:44 -0500
Subject: [PATCH 21/22] Simplify tests

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 .../CodeGen/AMDGPU/icache-prefetch-config.ll  |  78 +-----
 llvm/test/CodeGen/AMDGPU/icache-prefetch.ll   | 252 +-----------------
 2 files changed, 15 insertions(+), 315 deletions(-)

diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
index 3006a1ca588b3..0b5269712fb3a 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch-config.ll
@@ -1,62 +1,19 @@
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -o - %s | FileCheck -check-prefix=GFX1200 %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -o - %s | FileCheck -check-prefix=GFX1250 %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -o - %s | FileCheck -check-prefix=INITIAL-16K %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=32640 -o - %s | FileCheck -check-prefix=MAX-INITIAL %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=49152 -o - %s | FileCheck -check-prefix=THRESHOLD-48K %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=THRESHOLD-16K %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=EQUAL-16K %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=0 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=8193 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=32768 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-INITIAL %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=0 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=32769 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-threshold=65664 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-THRESHOLD %s
-; RUN: not --crash llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=8192 -o /dev/null %s 2>&1 | FileCheck -check-prefix=INVALID-ORDER %s
+; Check the gfx1200 default initial prefetch size.
+; RUN: llc -mtriple=amdgpu12.00-amd-amdhsa -o - %s | FileCheck -check-prefix=GFX1200 %s
+; Check overriding only the explicit-prefetch threshold.
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=THRESHOLD-16K %s
+; Check independently overriding the initial prefetch size.
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -amdgpu-icache-prefetch-initial-size=16384 -amdgpu-icache-prefetch-threshold=16384 -o - %s | FileCheck -check-prefix=EQUAL-16K %s
 
-; The default threshold is the 32640-byte descriptor capacity on both targets.
-; gfx1200 initially prefetches 16KiB, while gfx1250 initially prefetches 8KiB.
+; The default threshold is the 32640-byte descriptor capacity.
 define amdgpu_kernel void @above_default_threshold() {
 ; GFX1200-LABEL: above_default_threshold:
 ; GFX1200:       s_prefetch_inst_pc_rel prefetchoffset(128,
 ; GFX1200:       .amdhsa_inst_pref_size 128
-;
-; GFX1250-LABEL: above_default_threshold:
-; GFX1250:       s_prefetch_inst_pc_rel prefetchoffset(64,
-; GFX1250:       .amdhsa_inst_pref_size 64
-;
-; INITIAL-16K-LABEL: above_default_threshold:
-; INITIAL-16K:       s_prefetch_inst_pc_rel prefetchoffset(128,
-; INITIAL-16K:       .amdhsa_inst_pref_size 128
-;
-; MAX-INITIAL-LABEL: above_default_threshold:
-; MAX-INITIAL:       s_prefetch_inst_pc_rel prefetchoffset(255,
-; MAX-INITIAL:       .amdhsa_inst_pref_size 255
-;
-; THRESHOLD-48K-LABEL: above_default_threshold:
-; THRESHOLD-48K-NOT:   s_prefetch_inst_pc_rel
-; THRESHOLD-48K:       s_endpgm
-; THRESHOLD-48K:       .amdhsa_inst_pref_size ((instprefsize(
   call void asm sideeffect ".space 40000", ""()
   ret void
 }
 
-; The gfx1250 prologue and s_endpgm add 32 bytes, making these functions
-; exactly 32640 bytes and four bytes over 32640 bytes respectively.
-define amdgpu_kernel void @at_default_threshold() {
-; GFX1250-LABEL: at_default_threshold:
-; GFX1250-NOT:   s_prefetch_inst_pc_rel
-; GFX1250:       s_endpgm
-  call void asm sideeffect ".space 32608", ""()
-  ret void
-}
-
-define amdgpu_kernel void @above_default_threshold_by_four() {
-; GFX1250-LABEL: above_default_threshold_by_four:
-; GFX1250:       s_prefetch_inst_pc_rel prefetchoffset(64,
-  call void asm sideeffect ".space 32612", ""()
-  ret void
-}
-
 ; Lowering only the threshold activates explicit prefetching while retaining
 ; the default 8KiB initial descriptor coverage. Setting initial == threshold
 ; is also valid and changes the descriptor coverage independently.
@@ -71,24 +28,3 @@ define amdgpu_kernel void @between_override_thresholds() {
   call void asm sideeffect ".space 20000", ""()
   ret void
 }
-
-; INITIAL-16K-LABEL: cache_size_limit:
-; INITIAL-16K:          s_mov_b64
-; INITIAL-16K-NEXT:     v_nop
-; INITIAL-16K-NEXT:     global_prefetch_b8
-; INITIAL-16K-COUNT-12: s_prefetch_inst_pc_rel
-; INITIAL-16K-NEXT:     ;;#ASMSTART
-; GFX1250-LABEL:     cache_size_limit:
-; GFX1250:           s_mov_b64
-; GFX1250-NEXT:      v_nop
-; GFX1250-NEXT:      global_prefetch_b8
-; GFX1250-COUNT-14:  s_prefetch_inst_pc_rel
-; GFX1250-NEXT:      ;;#ASMSTART
-define amdgpu_kernel void @cache_size_limit() {
-  call void asm sideeffect ".space 65536", ""()
-  ret void
-}
-
-; INVALID-INITIAL: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must be a non-zero multiple of 128 bytes not exceeding 32640 bytes for gfx1250
-; INVALID-THRESHOLD: LLVM ERROR: -amdgpu-icache-prefetch-threshold must be a non-zero multiple of 128 bytes not exceeding 65536 bytes for gfx1250
-; INVALID-ORDER: LLVM ERROR: -amdgpu-icache-prefetch-initial-size must not exceed -amdgpu-icache-prefetch-threshold for gfx1250
diff --git a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
index 796702bda1c53..39c22d3d59722 100644
--- a/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/icache-prefetch.ll
@@ -1,18 +1,15 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -o - %s | \
+; Check prefetch insertion and distribution.
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -amdgpu-icache-prefetch=true -o - %s | \
 ; RUN:   FileCheck -check-prefix=GFX1250 %s
 
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=false -o - %s | \
-; RUN:   FileCheck -check-prefix=NO-PREFETCH %s
-
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -amdgpu-icache-prefetch-initial-size=16384 -o - %s | \
-; RUN:   FileCheck -check-prefix=GFX1250-DIST %s
-
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
+; Check final prefetch offsets and cache-line counts.
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -amdgpu-icache-prefetch=true -filetype=obj -o %t.o %s
 ; RUN: llvm-objdump -d %t.o | FileCheck -check-prefix=GFX1250-OBJ %s
 
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -amdgpu-icache-prefetch=true -stop-after=amdgpu-insert-icache-prefetch -o %t.mir %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1250 -start-after=amdgpu-insert-icache-prefetch -filetype=obj -o %t-resumed.o %t.mir
+; Check prefetch state survives a MIR round trip.
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -amdgpu-icache-prefetch=true -stop-after=amdgpu-insert-icache-prefetch -o %t.mir %s
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -start-after=amdgpu-insert-icache-prefetch -filetype=obj -o %t-resumed.o %t.mir
 ; RUN: llvm-objdump -d %t-resumed.o | FileCheck -check-prefix=GFX1250-OBJ %s
 
 ; Use .space to make the estimated MachineFunction size and final assembled
@@ -33,28 +30,6 @@ define amdgpu_kernel void @below_threshold() {
 ; GFX1250-NEXT:    .space 30000
 ; GFX1250-NEXT:    ;;#ASMEND
 ; GFX1250-NEXT:    s_endpgm
-;
-; NO-PREFETCH-LABEL: below_threshold:
-; NO-PREFETCH:       ; %bb.0:
-; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; NO-PREFETCH-NEXT:    s_mov_b64 s[64:65], 0
-; NO-PREFETCH-NEXT:    v_nop
-; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; NO-PREFETCH-NEXT:    ;;#ASMSTART
-; NO-PREFETCH-NEXT:    .space 30000
-; NO-PREFETCH-NEXT:    ;;#ASMEND
-; NO-PREFETCH-NEXT:    s_endpgm
-;
-; GFX1250-DIST-LABEL: below_threshold:
-; GFX1250-DIST:       ; %bb.0:
-; GFX1250-DIST-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-DIST-NEXT:    s_mov_b64 s[64:65], 0
-; GFX1250-DIST-NEXT:    v_nop
-; GFX1250-DIST-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX1250-DIST-NEXT:    ;;#ASMSTART
-; GFX1250-DIST-NEXT:    .space 30000
-; GFX1250-DIST-NEXT:    ;;#ASMEND
-; GFX1250-DIST-NEXT:    s_endpgm
   call void asm sideeffect ".space 30000", ""()
   ret void
 }
@@ -105,51 +80,6 @@ define amdgpu_kernel void @partial_final_slot(ptr addrspace(1) %out) {
 ; GFX1250-NEXT:    global_store_b32 v0, v0, s[0:1]
 ; GFX1250-NEXT:    s_endpgm
 ; GFX1250-NEXT:  .Lpref_func_end0:
-;
-; NO-PREFETCH-LABEL: partial_final_slot:
-; NO-PREFETCH:       ; %bb.0:
-; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; NO-PREFETCH-NEXT:    s_mov_b64 s[64:65], 0
-; NO-PREFETCH-NEXT:    v_nop
-; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; NO-PREFETCH-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
-; NO-PREFETCH-NEXT:    v_mov_b32_e32 v0, 0
-; NO-PREFETCH-NEXT:    ;;#ASMSTART
-; NO-PREFETCH-NEXT:    .space 40000
-; NO-PREFETCH-NEXT:    ;;#ASMEND
-; NO-PREFETCH-NEXT:    s_wait_kmcnt 0x0
-; NO-PREFETCH-NEXT:    global_store_b32 v0, v0, s[0:1]
-; NO-PREFETCH-NEXT:    s_endpgm
-;
-; GFX1250-DIST-LABEL: partial_final_slot:
-; GFX1250-DIST:       ; %bb.0:
-; GFX1250-DIST-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-DIST-NEXT:    s_mov_b64 s[64:65], 0
-; GFX1250-DIST-NEXT:    v_nop
-; GFX1250-DIST-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset0:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot), null, prefetchcachelines(128, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset0-partial_final_slot)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset1:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot), null, prefetchcachelines(160, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset1-partial_final_slot)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset2:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot), null, prefetchcachelines(192, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset2-partial_final_slot)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset3:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset3-partial_final_slot), null, prefetchcachelines(224, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset3-partial_final_slot)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset4:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset4-partial_final_slot), null, prefetchcachelines(256, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset4-partial_final_slot)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset5:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot), null, prefetchcachelines(288, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset5-partial_final_slot)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset6:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot), null, prefetchcachelines(320, .Lpref_func_end0-partial_final_slot, .Lpref_inst_offset6-partial_final_slot)
-; GFX1250-DIST-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
-; GFX1250-DIST-NEXT:    v_mov_b32_e32 v0, 0
-; GFX1250-DIST-NEXT:    ;;#ASMSTART
-; GFX1250-DIST-NEXT:    .space 40000
-; GFX1250-DIST-NEXT:    ;;#ASMEND
-; GFX1250-DIST-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-DIST-NEXT:    global_store_b32 v0, v0, s[0:1]
-; GFX1250-DIST-NEXT:    s_endpgm
-; GFX1250-DIST-NEXT:  .Lpref_func_end0:
   call void asm sideeffect ".space 40000", ""()
   store i32 0, ptr addrspace(1) %out
   ret void
@@ -212,58 +142,11 @@ define amdgpu_kernel void @cache_size_limit() {
 ; GFX1250-NEXT:    ;;#ASMEND
 ; GFX1250-NEXT:    s_endpgm
 ; GFX1250-NEXT:  .Lpref_func_end1:
-;
-; NO-PREFETCH-LABEL: cache_size_limit:
-; NO-PREFETCH:       ; %bb.0:
-; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; NO-PREFETCH-NEXT:    s_mov_b64 s[64:65], 0
-; NO-PREFETCH-NEXT:    v_nop
-; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; NO-PREFETCH-NEXT:    ;;#ASMSTART
-; NO-PREFETCH-NEXT:    .space 65536
-; NO-PREFETCH-NEXT:    ;;#ASMEND
-; NO-PREFETCH-NEXT:    s_endpgm
-;
-; GFX1250-DIST-LABEL: cache_size_limit:
-; GFX1250-DIST:       ; %bb.0:
-; GFX1250-DIST-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-DIST-NEXT:    s_mov_b64 s[64:65], 0
-; GFX1250-DIST-NEXT:    v_nop
-; GFX1250-DIST-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset7:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit), null, prefetchcachelines(128, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset7-cache_size_limit)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset8:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit), null, prefetchcachelines(160, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset8-cache_size_limit)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset9:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit), null, prefetchcachelines(192, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset9-cache_size_limit)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset10:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit), null, prefetchcachelines(224, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset10-cache_size_limit)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset11:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset11-cache_size_limit), null, prefetchcachelines(256, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset11-cache_size_limit)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset12:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset12-cache_size_limit), null, prefetchcachelines(288, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset12-cache_size_limit)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset13:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset13-cache_size_limit), null, prefetchcachelines(320, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset13-cache_size_limit)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset14:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset14-cache_size_limit), null, prefetchcachelines(352, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset14-cache_size_limit)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset15:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset15-cache_size_limit), null, prefetchcachelines(384, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset15-cache_size_limit)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset16:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset16-cache_size_limit), null, prefetchcachelines(416, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset16-cache_size_limit)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset17:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit), null, prefetchcachelines(448, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset17-cache_size_limit)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset18:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit), null, prefetchcachelines(480, .Lpref_func_end1-cache_size_limit, .Lpref_inst_offset18-cache_size_limit)
-; GFX1250-DIST-NEXT:    ;;#ASMSTART
-; GFX1250-DIST-NEXT:    .space 65536
-; GFX1250-DIST-NEXT:    ;;#ASMEND
-; GFX1250-DIST-NEXT:    s_endpgm
-; GFX1250-DIST-NEXT:  .Lpref_func_end1:
   call void asm sideeffect ".space 65536", ""()
   ret void
 }
 
-; With a 16KiB descriptor prefix override, explicit prefetches are deferred to
+; Explicit prefetches are distributed along the post-dominator chain through
 ; the shared exit block.
 declare i32 @llvm.amdgcn.workgroup.id.x()
 
@@ -331,107 +214,6 @@ define amdgpu_kernel void @postdominated_prefetch() {
 ; GFX1250-NEXT:    ;;#ASMEND
 ; GFX1250-NEXT:    s_endpgm
 ; GFX1250-NEXT:  .Lpref_func_end2:
-;
-; NO-PREFETCH-LABEL: postdominated_prefetch:
-; NO-PREFETCH:       ; %bb.0: ; %entry
-; NO-PREFETCH-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; NO-PREFETCH-NEXT:    s_mov_b64 s[64:65], 0
-; NO-PREFETCH-NEXT:    v_nop
-; NO-PREFETCH-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; NO-PREFETCH-NEXT:    s_bfe_u32 s0, ttmp6, 0x4000c
-; NO-PREFETCH-NEXT:    s_and_b32 s1, ttmp6, 15
-; NO-PREFETCH-NEXT:    s_add_co_i32 s0, s0, 1
-; NO-PREFETCH-NEXT:    s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
-; NO-PREFETCH-NEXT:    s_mul_i32 s0, ttmp9, s0
-; NO-PREFETCH-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
-; NO-PREFETCH-NEXT:    s_add_co_i32 s1, s1, s0
-; NO-PREFETCH-NEXT:    s_cmp_eq_u32 s2, 0
-; NO-PREFETCH-NEXT:    s_cselect_b32 s0, ttmp9, s1
-; NO-PREFETCH-NEXT:    s_cmp_lg_u32 s0, 0
-; NO-PREFETCH-NEXT:    s_mov_b32 s0, 0
-; NO-PREFETCH-NEXT:    s_cbranch_scc0 .LBB3_2
-; NO-PREFETCH-NEXT:  ; %bb.1: ; %else
-; NO-PREFETCH-NEXT:    ;;#ASMSTART
-; NO-PREFETCH-NEXT:    .space 8000
-; NO-PREFETCH-NEXT:    ;;#ASMEND
-; NO-PREFETCH-NEXT:    s_branch .LBB3_3
-; NO-PREFETCH-NEXT:  .LBB3_2:
-; NO-PREFETCH-NEXT:    s_mov_b32 s0, -1
-; NO-PREFETCH-NEXT:  .LBB3_3: ; %Flow
-; NO-PREFETCH-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
-; NO-PREFETCH-NEXT:    s_and_b32 s0, s0, exec_lo
-; NO-PREFETCH-NEXT:    s_cselect_b32 s0, 1, 0
-; NO-PREFETCH-NEXT:    s_cmp_lg_u32 s0, 1
-; NO-PREFETCH-NEXT:    s_cbranch_scc1 .LBB3_5
-; NO-PREFETCH-NEXT:  ; %bb.4: ; %then
-; NO-PREFETCH-NEXT:    ;;#ASMSTART
-; NO-PREFETCH-NEXT:    .space 8000
-; NO-PREFETCH-NEXT:    ;;#ASMEND
-; NO-PREFETCH-NEXT:  .LBB3_5: ; %join
-; NO-PREFETCH-NEXT:    ;;#ASMSTART
-; NO-PREFETCH-NEXT:    .space 32000
-; NO-PREFETCH-NEXT:    ;;#ASMEND
-; NO-PREFETCH-NEXT:    s_endpgm
-;
-; GFX1250-DIST-LABEL: postdominated_prefetch:
-; GFX1250-DIST:       ; %bb.0: ; %entry
-; GFX1250-DIST-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-DIST-NEXT:    s_mov_b64 s[64:65], 0
-; GFX1250-DIST-NEXT:    v_nop
-; GFX1250-DIST-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX1250-DIST-NEXT:    s_bfe_u32 s0, ttmp6, 0x4000c
-; GFX1250-DIST-NEXT:    s_and_b32 s1, ttmp6, 15
-; GFX1250-DIST-NEXT:    s_add_co_i32 s0, s0, 1
-; GFX1250-DIST-NEXT:    s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
-; GFX1250-DIST-NEXT:    s_mul_i32 s0, ttmp9, s0
-; GFX1250-DIST-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
-; GFX1250-DIST-NEXT:    s_add_co_i32 s1, s1, s0
-; GFX1250-DIST-NEXT:    s_cmp_eq_u32 s2, 0
-; GFX1250-DIST-NEXT:    s_cselect_b32 s0, ttmp9, s1
-; GFX1250-DIST-NEXT:    s_cmp_lg_u32 s0, 0
-; GFX1250-DIST-NEXT:    s_mov_b32 s0, 0
-; GFX1250-DIST-NEXT:    s_cbranch_scc0 .LBB3_2
-; GFX1250-DIST-NEXT:  ; %bb.1: ; %else
-; GFX1250-DIST-NEXT:    ;;#ASMSTART
-; GFX1250-DIST-NEXT:    .space 8000
-; GFX1250-DIST-NEXT:    ;;#ASMEND
-; GFX1250-DIST-NEXT:    s_branch .LBB3_3
-; GFX1250-DIST-NEXT:  .LBB3_2:
-; GFX1250-DIST-NEXT:    s_mov_b32 s0, -1
-; GFX1250-DIST-NEXT:  .LBB3_3: ; %Flow
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset19:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(128, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset19-postdominated_prefetch), null, prefetchcachelines(128, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset19-postdominated_prefetch)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset20:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(160, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset20-postdominated_prefetch), null, prefetchcachelines(160, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset20-postdominated_prefetch)
-; GFX1250-DIST-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
-; GFX1250-DIST-NEXT:    s_and_b32 s0, s0, exec_lo
-; GFX1250-DIST-NEXT:    s_cselect_b32 s0, 1, 0
-; GFX1250-DIST-NEXT:    s_cmp_lg_u32 s0, 1
-; GFX1250-DIST-NEXT:    s_cbranch_scc1 .LBB3_5
-; GFX1250-DIST-NEXT:  ; %bb.4: ; %then
-; GFX1250-DIST-NEXT:    ;;#ASMSTART
-; GFX1250-DIST-NEXT:    .space 8000
-; GFX1250-DIST-NEXT:    ;;#ASMEND
-; GFX1250-DIST-NEXT:  .LBB3_5: ; %join
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset21:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(192, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset21-postdominated_prefetch), null, prefetchcachelines(192, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset21-postdominated_prefetch)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset22:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(224, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset22-postdominated_prefetch), null, prefetchcachelines(224, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset22-postdominated_prefetch)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset23:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch), null, prefetchcachelines(256, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset23-postdominated_prefetch)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset24:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch), null, prefetchcachelines(288, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset24-postdominated_prefetch)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset25:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset25-postdominated_prefetch), null, prefetchcachelines(320, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset25-postdominated_prefetch)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset26:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset26-postdominated_prefetch), null, prefetchcachelines(352, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset26-postdominated_prefetch)
-; GFX1250-DIST-NEXT:  .Lpref_inst_offset27:
-; GFX1250-DIST-NEXT:    s_prefetch_inst_pc_rel prefetchoffset(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset27-postdominated_prefetch), null, prefetchcachelines(384, .Lpref_func_end2-postdominated_prefetch, .Lpref_inst_offset27-postdominated_prefetch)
-; GFX1250-DIST-NEXT:    ;;#ASMSTART
-; GFX1250-DIST-NEXT:    .space 32000
-; GFX1250-DIST-NEXT:    ;;#ASMEND
-; GFX1250-DIST-NEXT:    s_endpgm
-; GFX1250-DIST-NEXT:  .Lpref_func_end2:
 entry:
   %id = call i32 @llvm.amdgcn.workgroup.id.x()
   %cond = icmp eq i32 %id, 0
@@ -461,24 +243,6 @@ define void @called_function() {
 ; GFX1250-NEXT:    .space 40000
 ; GFX1250-NEXT:    ;;#ASMEND
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
-;
-; NO-PREFETCH-LABEL: called_function:
-; NO-PREFETCH:       ; %bb.0:
-; NO-PREFETCH-NEXT:    s_wait_loadcnt_dscnt 0x0
-; NO-PREFETCH-NEXT:    s_wait_kmcnt 0x0
-; NO-PREFETCH-NEXT:    ;;#ASMSTART
-; NO-PREFETCH-NEXT:    .space 40000
-; NO-PREFETCH-NEXT:    ;;#ASMEND
-; NO-PREFETCH-NEXT:    s_set_pc_i64 s[30:31]
-;
-; GFX1250-DIST-LABEL: called_function:
-; GFX1250-DIST:       ; %bb.0:
-; GFX1250-DIST-NEXT:    s_wait_loadcnt_dscnt 0x0
-; GFX1250-DIST-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-DIST-NEXT:    ;;#ASMSTART
-; GFX1250-DIST-NEXT:    .space 40000
-; GFX1250-DIST-NEXT:    ;;#ASMEND
-; GFX1250-DIST-NEXT:    s_set_pc_i64 s[30:31]
   call void asm sideeffect ".space 40000", ""()
   ret void
 }

>From f6056bd861735e8a65f9a2f1692d997f9ed0f0e2 Mon Sep 17 00:00:00 2001
From: Lukas Sommer <lukas.sommer at amd.com>
Date: Thu, 1 Oct 2026 03:48:28 -0500
Subject: [PATCH 22/22] Code formatting

Signed-off-by: Lukas Sommer <lukas.sommer at amd.com>
---
 llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h | 4 +---
 1 file changed, 1 insertion(+), 3 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
index 6e58ac3283318..e17f350ea2ef9 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
@@ -1137,9 +1137,7 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
 
   unsigned getICachePrefetchLines() const { return ICachePrefetchLines; }
 
-  void setICachePrefetchLines(unsigned Lines) {
-    ICachePrefetchLines = Lines;
-  }
+  void setICachePrefetchLines(unsigned Lines) { ICachePrefetchLines = Lines; }
 
   unsigned getNumSpilledSGPRs() const {
     return NumSpilledSGPRs;



More information about the llvm-commits mailing list