[llvm-branch-commits] [llvm] [AMDGPU] Fold provably out-of-bounds raw buffer loads/stores (PR #227402)

Dark Steve via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Thu Oct 1 04:23:19 PDT 2026


https://github.com/PrasoonMishra updated https://github.com/llvm/llvm-project/pull/227402

>From 8ae55c4a89e273b90d058d0044258b743979d7d4 Mon Sep 17 00:00:00 2001
From: Dark Steve Jobs <Prasoon.Mishra at amd.com>
Date: Tue, 29 Sep 2026 17:17:25 +0530
Subject: [PATCH 1/3] [AMDGPU] Fold provably out-of-bounds raw buffer
 loads/stores in InstCombine

Add an InstCombine fold for raw (IdxEn=0) buffer loads and stores. When
the resource descriptor is a compile-time constant and the access
offset can be proven to always fall outside its bounds, loads are
replaced with a zero constant and stores are erased.

Adds a BufferOOBModel subtarget feature to select the correct
range-check equation per hardware generation.

Scope:
- Only llvm.amdgcn.raw.ptr.buffer.load/store.
- Formatted/tbuffer loads, atomics, and struct (indexed) buffers are
  deferred to follow-ups.
---
 llvm/lib/Target/AMDGPU/AMDGPU.td              | 23 +++--
 .../AMDGPU/AMDGPUInstCombineIntrinsic.cpp     | 87 +++++++++++++++++++
 llvm/lib/Target/AMDGPU/GCNSubtarget.h         | 17 ++++
 3 files changed, 122 insertions(+), 5 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index f96934e0b8434..28dc132559828 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -1612,6 +1612,18 @@ class FeatureNumRecordsBufferResource<int width> : SubtargetFeature<
 def Feature32BitNumRecordsBufferResource : FeatureNumRecordsBufferResource<32>;
 def Feature45BitNumRecordsBufferResource : FeatureNumRecordsBufferResource<45>;
 
+// The out-of-bounds range-check model.
+class FeatureBufferOOBModel<string model, int value> : SubtargetFeature<
+  model#"-buffer-oob-model",
+  "BufferOOBModel",
+  !cast<string>(value),
+  "This target's hardware OOB range check follows the "#model#" model"
+>;
+
+def FeatureGfx9BufferOOBModel : FeatureBufferOOBModel<"Gfx9", 1>;
+def FeatureGfx10BufferOOBModel : FeatureBufferOOBModel<"Gfx10", 2>;
+def FeatureGfx1250BufferOOBModel : FeatureBufferOOBModel<"Gfx1250", 3>;
+
 defm Clusters : AMDGPUSubtargetFeature<"clusters",
   "Has clusters of workgroups support",
   /*GenPredicate=*/0
@@ -1758,7 +1770,7 @@ def FeatureGFX9 : GCNSubtargetFeatureGeneration<"GFX9",
    FeatureCvtNormInsts, FeatureCvtPkNormVOP2Insts,
    FeatureCvtPkNormVOP3Insts, FeatureDX10ClampAndIEEEMode,
    FeatureInstCacheLineSize64, FeatureMTBUFInsts, FeatureFormattedMUBUFInsts,
-   Feature32BitNumRecordsBufferResource
+   Feature32BitNumRecordsBufferResource, FeatureGfx9BufferOOBModel
   ]
 >;
 
@@ -1793,7 +1805,7 @@ def FeatureGFX10 : GCNSubtargetFeatureGeneration<"GFX10",
    FeatureCvtNormInsts, FeatureCvtPkNormVOP2Insts,
    FeatureCvtPkNormVOP3Insts, FeatureDX10ClampAndIEEEMode, FeatureFlatOffsetBits12,
    FeatureInstCacheLineSize64, FeatureMTBUFInsts, FeatureFormattedMUBUFInsts,
-   Feature32BitNumRecordsBufferResource
+   Feature32BitNumRecordsBufferResource, FeatureGfx10BufferOOBModel
   ]
 >;
 
@@ -1827,7 +1839,7 @@ def FeatureGFX11 : GCNSubtargetFeatureGeneration<"GFX11",
    FeatureMqsadInsts, FeatureCvtNormInsts,
    FeatureCvtPkNormVOP2Insts, FeatureCvtPkNormVOP3Insts,
    FeatureInstCacheLineSize128, FeatureMTBUFInsts, FeatureFormattedMUBUFInsts,
-   Feature32BitNumRecordsBufferResource,
+   Feature32BitNumRecordsBufferResource, FeatureGfx10BufferOOBModel,
    FeatureMaxWavesPerEU16
   ]
 >;
@@ -1888,7 +1900,7 @@ def FeatureGFX13 : GCNSubtargetFeatureGeneration<"GFX13",
    FeatureAgentScopeFineGrainedRemoteMemoryAtomics, FeatureFlatOffsetBits24,
    FeatureFlatSignedOffset, FeatureInstCacheLineSize128,
    FeatureMTBUFInsts, FeatureFormattedMUBUFInsts,
-   Feature45BitNumRecordsBufferResource,
+   Feature45BitNumRecordsBufferResource, FeatureGfx1250BufferOOBModel,
    FeatureMaxWavesPerEU16
   ]
 >;
@@ -2371,7 +2383,7 @@ def FeatureISAVersion11_7_Generic: FeatureSet<
 
 def FeatureISAVersion12 : FeatureSet<
   [FeatureGFX12,
-   Feature32BitNumRecordsBufferResource,
+   Feature32BitNumRecordsBufferResource, FeatureGfx10BufferOOBModel,
    FeatureSupportsWave64, FeatureSupportsWGP,
    FeatureBackOffBarrier,
    FeatureAddressableLocalMemorySize65536,
@@ -2520,6 +2532,7 @@ def FeatureISAVersion12_50_Common : FeatureSet<
    FeatureSetPrioIncWgInst,
    FeatureSWakeupBarrier,
    Feature45BitNumRecordsBufferResource,
+   FeatureGfx1250BufferOOBModel,
    FeatureClusters,
    FeatureD16Writes32BitVgpr,
    FeatureMcastLoadInsts,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
index 7721114fdd2f7..2f9299919d08c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
@@ -1175,6 +1175,79 @@ static Instruction *foldConstantIntoDotAccumulator(IntrinsicInst &II,
   return IC.replaceOperand(II, AccIdx, NewAcc);
 }
 
+// Check raw buffer access is OOB or not. For partial OOB access, just bail out.
+static bool isRawBufferAccessDefinitelyOOB(const GCNSubtarget *ST, Value *Rsrc,
+                                           Value *OffsetV, Value *SOffsetV,
+                                           Value *AuxV, uint64_t Payload,
+                                           const DataLayout &DL) {
+  using OOBModel = GCNSubtarget::AMDGPUBufferOOBModel;
+  std::optional<OOBModel> Model = ST->getBufferOOBModel();
+  std::optional<unsigned> NumRecordsWidth =
+      ST->getBufferResourceNumRecordsWidth();
+  if (!Model || !NumRecordsWidth)
+    return false;
+
+  auto *AuxC = dyn_cast<ConstantInt>(AuxV);
+  if (!AuxC)
+    return false;
+
+  uint64_t Offset = computeKnownBits(OffsetV, DL).getMinValue().getZExtValue();
+  uint64_t SOffset =
+      computeKnownBits(SOffsetV, DL).getMinValue().getZExtValue();
+
+  uint64_t Aux = AuxC->getZExtValue();
+  if (Aux & AMDGPU::CPol::VOLATILE)
+    return false;
+
+  // The resource descriptor must also be a compile-time constant.
+  const APInt *StrideAP, *NumRecAP, *FlagsAP;
+  if (!match(Rsrc, m_Intrinsic<Intrinsic::amdgcn_make_buffer_rsrc>(
+                       m_Value(), m_APInt(StrideAP), m_APInt(NumRecAP),
+                       m_APInt(FlagsAP))))
+    return false;
+
+  uint64_t StrideRaw = StrideAP->getZExtValue();
+  uint64_t Stride = StrideRaw & 0x3fff;
+  uint64_t Flags = FlagsAP->getZExtValue();
+  uint64_t NumRecordsMask = maskTrailingOnes<uint64_t>(*NumRecordsWidth);
+  uint64_t NumRecords =
+      NumRecAP->zextOrTrunc(64).getZExtValue() & NumRecordsMask;
+
+  switch (*Model) {
+  case OOBModel::Gfx9:
+  case OOBModel::Gfx10: {
+    bool AddTid = (Flags >> 23) & 1;
+    bool Swizzle = AMDGPU::isGFX11Plus(*ST) ? (StrideRaw >> 14) != 0
+                                            : (StrideRaw >> 15) != 0;
+    if (AddTid || Swizzle)
+      return false;
+
+    uint64_t Bound = NumRecords > SOffset ? NumRecords - SOffset : 0;
+    if (*Model == OOBModel::Gfx10) {
+      unsigned OOBSelect = (Flags >> 28) & 3;
+      return OOBSelect == 3 && Offset + Payload > Bound;
+    }
+    // Nonzero stride drops soffset out of the gfx9 mode equation.
+    return Offset + Payload > (Stride == 0 ? Bound : NumRecords);
+  }
+  case OOBModel::Gfx1250: {
+    bool IsBuffer = ((Flags >> 2) & 3) == 0;
+    bool Swizzle = Flags & 1;
+    bool StrideScaled = (StrideRaw >> 14) != 0;
+    bool OOBCheckDisabled = NumRecords == NumRecordsMask || NumRecords == 0;
+    if (!IsBuffer || Swizzle || StrideScaled || OOBCheckDisabled)
+      return false;
+
+    unsigned OOBSelect = (Flags >> 1) & 1;
+    // Here SOffset is added directly into the offset used for bound check.
+    uint64_t Addr = Offset + SOffset;
+    return Addr + Payload > NumRecords ||
+           (OOBSelect == 1 && Addr + Payload > Stride);
+  }
+  }
+  return false;
+}
+
 std::optional<Instruction *>
 GCNTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
   Intrinsic::ID IID = II.getIntrinsicID();
@@ -2104,6 +2177,20 @@ GCNTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
       return IC.replaceInstUsesWith(II, PoisonValue::get(II.getType()));
     return std::nullopt;
   }
+  case Intrinsic::amdgcn_raw_ptr_buffer_load: {
+    if (!isRawBufferAccessDefinitelyOOB(
+            ST, II.getArgOperand(0), II.getArgOperand(1), II.getArgOperand(2),
+            II.getArgOperand(3), /*Payload=*/1, IC.getDataLayout()))
+      break;
+    return IC.replaceInstUsesWith(II, Constant::getNullValue(II.getType()));
+  }
+  case Intrinsic::amdgcn_raw_ptr_buffer_store: {
+    if (!isRawBufferAccessDefinitelyOOB(
+            ST, II.getArgOperand(1), II.getArgOperand(2), II.getArgOperand(3),
+            II.getArgOperand(4), /*Payload=*/1, IC.getDataLayout()))
+      break;
+    return IC.eraseInstFromFunction(II);
+  }
   case Intrinsic::amdgcn_raw_buffer_store_format:
   case Intrinsic::amdgcn_struct_buffer_store_format:
   case Intrinsic::amdgcn_raw_tbuffer_store:
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 2e09f1f1fda39..54d89462d9368 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -59,6 +59,12 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
     LLVMAMDHSADebugTrap = 0x03,
   };
 
+  enum class AMDGPUBufferOOBModel {
+    Gfx9 = 1,
+    Gfx10 = 2,
+    Gfx1250 = 3,
+  };
+
 private:
   /// SelectionDAGISel related APIs.
   std::unique_ptr<const SelectionDAGTargetInfo> TSInfo;
@@ -88,6 +94,9 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
   /// set from tablegen subtarget features, 0 is unknown.
   unsigned BufferResourceNumRecordsWidth = 0;
 
+  /// OOB range-check model, set from tablegen subtarget features. 0 is unknown.
+  unsigned BufferOOBModel = 0;
+
   // Dynamically set bits that enable features.
   bool ScalarizeGlobal = false;
   const bool BufferOOBRelaxed;
@@ -357,6 +366,14 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
     return BufferResourceNumRecordsWidth;
   }
 
+  /// Return the OOB range-check model this hardware implements or nullopt if
+  /// unknown.
+  std::optional<AMDGPUBufferOOBModel> getBufferOOBModel() const {
+    if (BufferOOBModel == 0)
+      return std::nullopt;
+    return static_cast<AMDGPUBufferOOBModel>(BufferOOBModel);
+  }
+
   bool isCuModeEnabled() const { return EnableCuMode; }
 
   /// \returns Whether a work-group runs on all of the block's SIMDs.

>From eda4560b5687b82b70b3064ace866deb5f22db8f Mon Sep 17 00:00:00 2001
From: Dark Steve Jobs <Prasoon.Mishra at amd.com>
Date: Tue, 29 Sep 2026 22:18:10 +0530
Subject: [PATCH 2/3]  Add tests.

---
 .../InstCombine/AMDGPU/fold-oob-raw-buffer.ll | 250 ++++--------------
 1 file changed, 58 insertions(+), 192 deletions(-)

diff --git a/llvm/test/Transforms/InstCombine/AMDGPU/fold-oob-raw-buffer.ll b/llvm/test/Transforms/InstCombine/AMDGPU/fold-oob-raw-buffer.ll
index fd7050d4854fb..19fa48a02d81d 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/fold-oob-raw-buffer.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/fold-oob-raw-buffer.ll
@@ -12,9 +12,7 @@
 define i32 @gfx9_stride_zero_soffset_subtracted_fold(ptr %p) {
 ; GFX9-LABEL: define i32 @gfx9_stride_zero_soffset_subtracted_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 8, i32 8, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @gfx9_stride_zero_soffset_subtracted_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
@@ -36,15 +34,11 @@ define i32 @gfx9_stride_zero_soffset_subtracted_fold(ptr %p) {
 ;
 ; GFX1250-LABEL: define i32 @gfx9_stride_zero_soffset_subtracted_fold(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX1250-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 8, i32 8, i32 0)
-; GFX1250-NEXT:    ret i32 [[V]]
+; GFX1250-NEXT:    ret i32 0
 ;
 ; GFX13-LABEL: define i32 @gfx9_stride_zero_soffset_subtracted_fold(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX13-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 8, i32 8, i32 0)
-; GFX13-NEXT:    ret i32 [[V]]
+; GFX13-NEXT:    ret i32 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 0)
   %v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 8, i32 8, i32 0)
@@ -55,9 +49,7 @@ define i32 @gfx9_stride_zero_soffset_subtracted_fold(ptr %p) {
 define i32 @gfx9_stride_nonzero_soffset_ignored_fold(ptr %p) {
 ; GFX9-LABEL: define i32 @gfx9_stride_nonzero_soffset_ignored_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 4, i64 16, i32 0)
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 16, i32 100, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @gfx9_stride_nonzero_soffset_ignored_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -79,15 +71,11 @@ define i32 @gfx9_stride_nonzero_soffset_ignored_fold(ptr %p) {
 ;
 ; GFX1250-LABEL: define i32 @gfx9_stride_nonzero_soffset_ignored_fold(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 4, i64 16, i32 0)
-; GFX1250-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 16, i32 100, i32 0)
-; GFX1250-NEXT:    ret i32 [[V]]
+; GFX1250-NEXT:    ret i32 0
 ;
 ; GFX13-LABEL: define i32 @gfx9_stride_nonzero_soffset_ignored_fold(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 4, i64 16, i32 0)
-; GFX13-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 16, i32 100, i32 0)
-; GFX13-NEXT:    ret i32 [[V]]
+; GFX13-NEXT:    ret i32 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 4, i64 16, i32 0)
   %v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 16, i32 100, i32 0)
@@ -122,15 +110,11 @@ define i32 @gfx9_addtid_no_fold(ptr %p) {
 ;
 ; GFX1250-LABEL: define i32 @gfx9_addtid_no_fold(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 8388608)
-; GFX1250-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT:    ret i32 [[V]]
+; GFX1250-NEXT:    ret i32 0
 ;
 ; GFX13-LABEL: define i32 @gfx9_addtid_no_fold(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 8388608)
-; GFX13-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT:    ret i32 [[V]]
+; GFX13-NEXT:    ret i32 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 8388608)
   %v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -227,9 +211,7 @@ define i32 @gfx9_volatile_no_fold(ptr %p) {
 define i32 @gfx9_boundary_offset_equals_bound_fold(ptr %p) {
 ; GFX9-LABEL: define i32 @gfx9_boundary_offset_equals_bound_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 64, i32 0)
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 64, i32 0, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @gfx9_boundary_offset_equals_bound_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -251,15 +233,11 @@ define i32 @gfx9_boundary_offset_equals_bound_fold(ptr %p) {
 ;
 ; GFX1250-LABEL: define i32 @gfx9_boundary_offset_equals_bound_fold(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 64, i32 0)
-; GFX1250-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 64, i32 0, i32 0)
-; GFX1250-NEXT:    ret i32 [[V]]
+; GFX1250-NEXT:    ret i32 0
 ;
 ; GFX13-LABEL: define i32 @gfx9_boundary_offset_equals_bound_fold(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 64, i32 0)
-; GFX13-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 64, i32 0, i32 0)
-; GFX13-NEXT:    ret i32 [[V]]
+; GFX13-NEXT:    ret i32 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 64, i32 0)
   %v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 64, i32 0, i32 0)
@@ -315,39 +293,27 @@ define i32 @gfx9_boundary_offset_below_bound_no_fold(ptr %p) {
 define i32 @gfx10_oobselect3_fold(ptr %p) {
 ; GFX9-LABEL: define i32 @gfx10_oobselect3_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @gfx10_oobselect3_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX10-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX10-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX10-NEXT:    ret i32 [[V]]
+; GFX10-NEXT:    ret i32 0
 ;
 ; GFX11-LABEL: define i32 @gfx10_oobselect3_fold(
 ; GFX11-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX11-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX11-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX11-NEXT:    ret i32 [[V]]
+; GFX11-NEXT:    ret i32 0
 ;
 ; GFX12-LABEL: define i32 @gfx10_oobselect3_fold(
 ; GFX12-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX12-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX12-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX12-NEXT:    ret i32 [[V]]
+; GFX12-NEXT:    ret i32 0
 ;
 ; GFX1250-LABEL: define i32 @gfx10_oobselect3_fold(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX1250-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT:    ret i32 [[V]]
+; GFX1250-NEXT:    ret i32 0
 ;
 ; GFX13-LABEL: define i32 @gfx10_oobselect3_fold(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX13-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT:    ret i32 [[V]]
+; GFX13-NEXT:    ret i32 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 805306368)
   %v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -358,9 +324,7 @@ define i32 @gfx10_oobselect3_fold(ptr %p) {
 define i32 @gfx10_oobselect0_no_fold(ptr %p) {
 ; GFX9-LABEL: define i32 @gfx10_oobselect0_no_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @gfx10_oobselect0_no_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -382,15 +346,11 @@ define i32 @gfx10_oobselect0_no_fold(ptr %p) {
 ;
 ; GFX1250-LABEL: define i32 @gfx10_oobselect0_no_fold(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX1250-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT:    ret i32 [[V]]
+; GFX1250-NEXT:    ret i32 0
 ;
 ; GFX13-LABEL: define i32 @gfx10_oobselect0_no_fold(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX13-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT:    ret i32 [[V]]
+; GFX13-NEXT:    ret i32 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 0)
   %v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -425,15 +385,11 @@ define i32 @gfx10_addtid_no_fold(ptr %p) {
 ;
 ; GFX1250-LABEL: define i32 @gfx10_addtid_no_fold(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 813694976)
-; GFX1250-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT:    ret i32 [[V]]
+; GFX1250-NEXT:    ret i32 0
 ;
 ; GFX13-LABEL: define i32 @gfx10_addtid_no_fold(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 813694976)
-; GFX13-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT:    ret i32 [[V]]
+; GFX13-NEXT:    ret i32 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 813694976)
   %v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -444,39 +400,27 @@ define i32 @gfx10_addtid_no_fold(ptr %p) {
 define i32 @gfx12_swizzle_disabled_fold(ptr %p) {
 ; GFX9-LABEL: define i32 @gfx12_swizzle_disabled_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @gfx12_swizzle_disabled_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX10-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX10-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX10-NEXT:    ret i32 [[V]]
+; GFX10-NEXT:    ret i32 0
 ;
 ; GFX11-LABEL: define i32 @gfx12_swizzle_disabled_fold(
 ; GFX11-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX11-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX11-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX11-NEXT:    ret i32 [[V]]
+; GFX11-NEXT:    ret i32 0
 ;
 ; GFX12-LABEL: define i32 @gfx12_swizzle_disabled_fold(
 ; GFX12-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX12-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX12-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX12-NEXT:    ret i32 [[V]]
+; GFX12-NEXT:    ret i32 0
 ;
 ; GFX1250-LABEL: define i32 @gfx12_swizzle_disabled_fold(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX1250-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT:    ret i32 [[V]]
+; GFX1250-NEXT:    ret i32 0
 ;
 ; GFX13-LABEL: define i32 @gfx12_swizzle_disabled_fold(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX13-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT:    ret i32 [[V]]
+; GFX13-NEXT:    ret i32 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 805306368)
   %v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -489,9 +433,7 @@ define i32 @gfx12_swizzle_disabled_fold(ptr %p) {
 define i32 @gfx1250_nonbuffer_type_no_fold(ptr %p) {
 ; GFX9-LABEL: define i32 @gfx1250_nonbuffer_type_no_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 4)
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @gfx1250_nonbuffer_type_no_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -533,9 +475,7 @@ define i32 @gfx1250_nonbuffer_type_no_fold(ptr %p) {
 define i32 @gfx1250_stride_scaled_no_fold(ptr %p) {
 ; GFX9-LABEL: define i32 @gfx1250_stride_scaled_no_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 16384, i64 1000000, i32 0)
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 2000000000, i32 0, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @gfx1250_stride_scaled_no_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -576,9 +516,7 @@ define i32 @gfx1250_stride_scaled_no_fold(ptr %p) {
 define i32 @gfx1250_numrecords_zero_no_fold(ptr %p) {
 ; GFX9-LABEL: define i32 @gfx1250_numrecords_zero_no_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 0, i32 0)
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 100, i32 0, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @gfx1250_numrecords_zero_no_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -620,9 +558,7 @@ define i32 @gfx1250_numrecords_zero_no_fold(ptr %p) {
 define i32 @gfx1250_numrecords_allones_no_fold(ptr %p) {
 ; GFX9-LABEL: define i32 @gfx1250_numrecords_allones_no_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 35184372088831, i32 0)
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 -1, i32 0, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @gfx1250_numrecords_allones_no_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -663,9 +599,7 @@ define i32 @gfx1250_numrecords_allones_no_fold(ptr %p) {
 define i32 @gfx1250_oobselect0_fold(ptr %p) {
 ; GFX9-LABEL: define i32 @gfx1250_oobselect0_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @gfx1250_oobselect0_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -687,15 +621,11 @@ define i32 @gfx1250_oobselect0_fold(ptr %p) {
 ;
 ; GFX1250-LABEL: define i32 @gfx1250_oobselect0_fold(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX1250-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT:    ret i32 [[V]]
+; GFX1250-NEXT:    ret i32 0
 ;
 ; GFX13-LABEL: define i32 @gfx1250_oobselect0_fold(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX13-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT:    ret i32 [[V]]
+; GFX13-NEXT:    ret i32 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 0)
   %v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -730,15 +660,11 @@ define i32 @gfx1250_oobselect1_stride0_always_oob(ptr %p) {
 ;
 ; GFX1250-LABEL: define i32 @gfx1250_oobselect1_stride0_always_oob(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 1000000, i32 2)
-; GFX1250-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 0, i32 0, i32 0)
-; GFX1250-NEXT:    ret i32 [[V]]
+; GFX1250-NEXT:    ret i32 0
 ;
 ; GFX13-LABEL: define i32 @gfx1250_oobselect1_stride0_always_oob(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 1000000, i32 2)
-; GFX13-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 0, i32 0, i32 0)
-; GFX13-NEXT:    ret i32 [[V]]
+; GFX13-NEXT:    ret i32 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 1000000, i32 2)
   %v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 0, i32 0, i32 0)
@@ -750,9 +676,7 @@ define i32 @gfx1250_oobselect1_stride0_always_oob(ptr %p) {
 define i32 @gfx1250_soffset_alone_sufficient_fold(ptr %p) {
 ; GFX9-LABEL: define i32 @gfx1250_soffset_alone_sufficient_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 0, i32 32, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @gfx1250_soffset_alone_sufficient_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -774,15 +698,11 @@ define i32 @gfx1250_soffset_alone_sufficient_fold(ptr %p) {
 ;
 ; GFX1250-LABEL: define i32 @gfx1250_soffset_alone_sufficient_fold(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX1250-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 0, i32 32, i32 0)
-; GFX1250-NEXT:    ret i32 [[V]]
+; GFX1250-NEXT:    ret i32 0
 ;
 ; GFX13-LABEL: define i32 @gfx1250_soffset_alone_sufficient_fold(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX13-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 0, i32 32, i32 0)
-; GFX13-NEXT:    ret i32 [[V]]
+; GFX13-NEXT:    ret i32 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 0)
   %v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 0, i32 32, i32 0)
@@ -795,45 +715,27 @@ define i32 @gfx1250_soffset_alone_sufficient_fold(ptr %p) {
 define i32 @known_bits_offset_provable_fold(ptr %p, i32 %x) {
 ; GFX9-LABEL: define i32 @known_bits_offset_provable_fold(
 ; GFX9-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX9-NEXT:    [[OFF:%.*]] = or i32 [[X]], 2048
-; GFX9-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 [[OFF]], i32 0, i32 0)
-; GFX9-NEXT:    ret i32 [[V]]
+; GFX9-NEXT:    ret i32 0
 ;
 ; GFX10-LABEL: define i32 @known_bits_offset_provable_fold(
 ; GFX10-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) #[[ATTR0]] {
-; GFX10-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX10-NEXT:    [[OFF:%.*]] = or i32 [[X]], 2048
-; GFX10-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 [[OFF]], i32 0, i32 0)
-; GFX10-NEXT:    ret i32 [[V]]
+; GFX10-NEXT:    ret i32 0
 ;
 ; GFX11-LABEL: define i32 @known_bits_offset_provable_fold(
 ; GFX11-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) #[[ATTR0]] {
-; GFX11-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX11-NEXT:    [[OFF:%.*]] = or i32 [[X]], 2048
-; GFX11-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 [[OFF]], i32 0, i32 0)
-; GFX11-NEXT:    ret i32 [[V]]
+; GFX11-NEXT:    ret i32 0
 ;
 ; GFX12-LABEL: define i32 @known_bits_offset_provable_fold(
 ; GFX12-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) #[[ATTR0]] {
-; GFX12-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX12-NEXT:    [[OFF:%.*]] = or i32 [[X]], 2048
-; GFX12-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 [[OFF]], i32 0, i32 0)
-; GFX12-NEXT:    ret i32 [[V]]
+; GFX12-NEXT:    ret i32 0
 ;
 ; GFX1250-LABEL: define i32 @known_bits_offset_provable_fold(
 ; GFX1250-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX1250-NEXT:    [[OFF:%.*]] = or i32 [[X]], 2048
-; GFX1250-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 [[OFF]], i32 0, i32 0)
-; GFX1250-NEXT:    ret i32 [[V]]
+; GFX1250-NEXT:    ret i32 0
 ;
 ; GFX13-LABEL: define i32 @known_bits_offset_provable_fold(
 ; GFX13-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX13-NEXT:    [[OFF:%.*]] = or i32 [[X]], 2048
-; GFX13-NEXT:    [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 [[OFF]], i32 0, i32 0)
-; GFX13-NEXT:    ret i32 [[V]]
+; GFX13-NEXT:    ret i32 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 805306368)
   %off = or i32 %x, 2048
@@ -940,39 +842,27 @@ define i32 @known_bits_neither_sufficient_no_fold(ptr %p, i32 %x, i32 %y) {
 define <4 x i32> @wide_vector_load_fold(ptr %p) {
 ; GFX9-LABEL: define <4 x i32> @wide_vector_load_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX9-NEXT:    [[V:%.*]] = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT:    ret <4 x i32> [[V]]
+; GFX9-NEXT:    ret <4 x i32> zeroinitializer
 ;
 ; GFX10-LABEL: define <4 x i32> @wide_vector_load_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX10-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX10-NEXT:    [[V:%.*]] = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX10-NEXT:    ret <4 x i32> [[V]]
+; GFX10-NEXT:    ret <4 x i32> zeroinitializer
 ;
 ; GFX11-LABEL: define <4 x i32> @wide_vector_load_fold(
 ; GFX11-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX11-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX11-NEXT:    [[V:%.*]] = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX11-NEXT:    ret <4 x i32> [[V]]
+; GFX11-NEXT:    ret <4 x i32> zeroinitializer
 ;
 ; GFX12-LABEL: define <4 x i32> @wide_vector_load_fold(
 ; GFX12-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX12-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX12-NEXT:    [[V:%.*]] = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX12-NEXT:    ret <4 x i32> [[V]]
+; GFX12-NEXT:    ret <4 x i32> zeroinitializer
 ;
 ; GFX1250-LABEL: define <4 x i32> @wide_vector_load_fold(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX1250-NEXT:    [[V:%.*]] = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT:    ret <4 x i32> [[V]]
+; GFX1250-NEXT:    ret <4 x i32> zeroinitializer
 ;
 ; GFX13-LABEL: define <4 x i32> @wide_vector_load_fold(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX13-NEXT:    [[V:%.*]] = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT:    ret <4 x i32> [[V]]
+; GFX13-NEXT:    ret <4 x i32> zeroinitializer
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 805306368)
   %v = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -983,39 +873,27 @@ define <4 x i32> @wide_vector_load_fold(ptr %p) {
 define i64 @wide_i64_load_fold(ptr %p) {
 ; GFX9-LABEL: define i64 @wide_i64_load_fold(
 ; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX9-NEXT:    [[V:%.*]] = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT:    ret i64 [[V]]
+; GFX9-NEXT:    ret i64 0
 ;
 ; GFX10-LABEL: define i64 @wide_i64_load_fold(
 ; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX10-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX10-NEXT:    [[V:%.*]] = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX10-NEXT:    ret i64 [[V]]
+; GFX10-NEXT:    ret i64 0
 ;
 ; GFX11-LABEL: define i64 @wide_i64_load_fold(
 ; GFX11-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX11-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX11-NEXT:    [[V:%.*]] = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX11-NEXT:    ret i64 [[V]]
+; GFX11-NEXT:    ret i64 0
 ;
 ; GFX12-LABEL: define i64 @wide_i64_load_fold(
 ; GFX12-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX12-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX12-NEXT:    [[V:%.*]] = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX12-NEXT:    ret i64 [[V]]
+; GFX12-NEXT:    ret i64 0
 ;
 ; GFX1250-LABEL: define i64 @wide_i64_load_fold(
 ; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX1250-NEXT:    [[V:%.*]] = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT:    ret i64 [[V]]
+; GFX1250-NEXT:    ret i64 0
 ;
 ; GFX13-LABEL: define i64 @wide_i64_load_fold(
 ; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX13-NEXT:    [[V:%.*]] = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT:    ret i64 [[V]]
+; GFX13-NEXT:    ret i64 0
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 805306368)
   %v = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -1026,38 +904,26 @@ define i64 @wide_i64_load_fold(ptr %p) {
 define void @wide_vector_store_erased(ptr %p, <4 x i32> %val) {
 ; GFX9-LABEL: define void @wide_vector_store_erased(
 ; GFX9-SAME: ptr [[P:%.*]], <4 x i32> [[VAL:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX9-NEXT:    call void @llvm.amdgcn.raw.ptr.buffer.store.v4i32(<4 x i32> [[VAL]], ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
 ; GFX9-NEXT:    ret void
 ;
 ; GFX10-LABEL: define void @wide_vector_store_erased(
 ; GFX10-SAME: ptr [[P:%.*]], <4 x i32> [[VAL:%.*]]) #[[ATTR0]] {
-; GFX10-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX10-NEXT:    call void @llvm.amdgcn.raw.ptr.buffer.store.v4i32(<4 x i32> [[VAL]], ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
 ; GFX10-NEXT:    ret void
 ;
 ; GFX11-LABEL: define void @wide_vector_store_erased(
 ; GFX11-SAME: ptr [[P:%.*]], <4 x i32> [[VAL:%.*]]) #[[ATTR0]] {
-; GFX11-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX11-NEXT:    call void @llvm.amdgcn.raw.ptr.buffer.store.v4i32(<4 x i32> [[VAL]], ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
 ; GFX11-NEXT:    ret void
 ;
 ; GFX12-LABEL: define void @wide_vector_store_erased(
 ; GFX12-SAME: ptr [[P:%.*]], <4 x i32> [[VAL:%.*]]) #[[ATTR0]] {
-; GFX12-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX12-NEXT:    call void @llvm.amdgcn.raw.ptr.buffer.store.v4i32(<4 x i32> [[VAL]], ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
 ; GFX12-NEXT:    ret void
 ;
 ; GFX1250-LABEL: define void @wide_vector_store_erased(
 ; GFX1250-SAME: ptr [[P:%.*]], <4 x i32> [[VAL:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX1250-NEXT:    call void @llvm.amdgcn.raw.ptr.buffer.store.v4i32(<4 x i32> [[VAL]], ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
 ; GFX1250-NEXT:    ret void
 ;
 ; GFX13-LABEL: define void @wide_vector_store_erased(
 ; GFX13-SAME: ptr [[P:%.*]], <4 x i32> [[VAL:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT:    [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX13-NEXT:    call void @llvm.amdgcn.raw.ptr.buffer.store.v4i32(<4 x i32> [[VAL]], ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
 ; GFX13-NEXT:    ret void
 ;
   %rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 805306368)

>From 418d68df1c3b52d0cf30f3f83ebd2830279f4272 Mon Sep 17 00:00:00 2001
From: Dark Steve Jobs <Prasoon.Mishra at amd.com>
Date: Thu, 1 Oct 2026 14:18:39 +0530
Subject: [PATCH 3/3] Addressed the comments.

---
 llvm/lib/Target/AMDGPU/AMDGPU.td                      | 5 ++++-
 llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp | 1 +
 2 files changed, 5 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index 28dc132559828..3487e08371890 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -1612,7 +1612,10 @@ class FeatureNumRecordsBufferResource<int width> : SubtargetFeature<
 def Feature32BitNumRecordsBufferResource : FeatureNumRecordsBufferResource<32>;
 def Feature45BitNumRecordsBufferResource : FeatureNumRecordsBufferResource<45>;
 
-// The out-of-bounds range-check model.
+// The out-of-bounds range-check model. The buffer resource (V#) has no
+// OOB_SELECT field on Gfx9, a 2-bit OOB_SELECT on Gfx10 (gfx10-gfx12),
+// and a 1-bit OOB_SELECT on Gfx1250 (gfx1250+). The OOB range-check
+// equations differ per model.
 class FeatureBufferOOBModel<string model, int value> : SubtargetFeature<
   model#"-buffer-oob-model",
   "BufferOOBModel",
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
index 2f9299919d08c..5e3c2242a3311 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
@@ -1222,6 +1222,7 @@ static bool isRawBufferAccessDefinitelyOOB(const GCNSubtarget *ST, Value *Rsrc,
     if (AddTid || Swizzle)
       return false;
 
+    // Saturating uint64 subtract: clamp at 0 if SOffset >= NumRecords.
     uint64_t Bound = NumRecords > SOffset ? NumRecords - SOffset : 0;
     if (*Model == OOBModel::Gfx10) {
       unsigned OOBSelect = (Flags >> 28) & 3;



More information about the llvm-branch-commits mailing list