[llvm-branch-commits] [llvm] [AMDGPU] Fold provably out-of-bounds raw buffer loads/stores (PR #227402)
Dark Steve via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Thu Oct 1 04:23:19 PDT 2026
https://github.com/PrasoonMishra updated https://github.com/llvm/llvm-project/pull/227402
>From 8ae55c4a89e273b90d058d0044258b743979d7d4 Mon Sep 17 00:00:00 2001
From: Dark Steve Jobs <Prasoon.Mishra at amd.com>
Date: Tue, 29 Sep 2026 17:17:25 +0530
Subject: [PATCH 1/3] [AMDGPU] Fold provably out-of-bounds raw buffer
loads/stores in InstCombine
Add an InstCombine fold for raw (IdxEn=0) buffer loads and stores. When
the resource descriptor is a compile-time constant and the access
offset can be proven to always fall outside its bounds, loads are
replaced with a zero constant and stores are erased.
Adds a BufferOOBModel subtarget feature to select the correct
range-check equation per hardware generation.
Scope:
- Only llvm.amdgcn.raw.ptr.buffer.load/store.
- Formatted/tbuffer loads, atomics, and struct (indexed) buffers are
deferred to follow-ups.
---
llvm/lib/Target/AMDGPU/AMDGPU.td | 23 +++--
.../AMDGPU/AMDGPUInstCombineIntrinsic.cpp | 87 +++++++++++++++++++
llvm/lib/Target/AMDGPU/GCNSubtarget.h | 17 ++++
3 files changed, 122 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index f96934e0b8434..28dc132559828 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -1612,6 +1612,18 @@ class FeatureNumRecordsBufferResource<int width> : SubtargetFeature<
def Feature32BitNumRecordsBufferResource : FeatureNumRecordsBufferResource<32>;
def Feature45BitNumRecordsBufferResource : FeatureNumRecordsBufferResource<45>;
+// The out-of-bounds range-check model.
+class FeatureBufferOOBModel<string model, int value> : SubtargetFeature<
+ model#"-buffer-oob-model",
+ "BufferOOBModel",
+ !cast<string>(value),
+ "This target's hardware OOB range check follows the "#model#" model"
+>;
+
+def FeatureGfx9BufferOOBModel : FeatureBufferOOBModel<"Gfx9", 1>;
+def FeatureGfx10BufferOOBModel : FeatureBufferOOBModel<"Gfx10", 2>;
+def FeatureGfx1250BufferOOBModel : FeatureBufferOOBModel<"Gfx1250", 3>;
+
defm Clusters : AMDGPUSubtargetFeature<"clusters",
"Has clusters of workgroups support",
/*GenPredicate=*/0
@@ -1758,7 +1770,7 @@ def FeatureGFX9 : GCNSubtargetFeatureGeneration<"GFX9",
FeatureCvtNormInsts, FeatureCvtPkNormVOP2Insts,
FeatureCvtPkNormVOP3Insts, FeatureDX10ClampAndIEEEMode,
FeatureInstCacheLineSize64, FeatureMTBUFInsts, FeatureFormattedMUBUFInsts,
- Feature32BitNumRecordsBufferResource
+ Feature32BitNumRecordsBufferResource, FeatureGfx9BufferOOBModel
]
>;
@@ -1793,7 +1805,7 @@ def FeatureGFX10 : GCNSubtargetFeatureGeneration<"GFX10",
FeatureCvtNormInsts, FeatureCvtPkNormVOP2Insts,
FeatureCvtPkNormVOP3Insts, FeatureDX10ClampAndIEEEMode, FeatureFlatOffsetBits12,
FeatureInstCacheLineSize64, FeatureMTBUFInsts, FeatureFormattedMUBUFInsts,
- Feature32BitNumRecordsBufferResource
+ Feature32BitNumRecordsBufferResource, FeatureGfx10BufferOOBModel
]
>;
@@ -1827,7 +1839,7 @@ def FeatureGFX11 : GCNSubtargetFeatureGeneration<"GFX11",
FeatureMqsadInsts, FeatureCvtNormInsts,
FeatureCvtPkNormVOP2Insts, FeatureCvtPkNormVOP3Insts,
FeatureInstCacheLineSize128, FeatureMTBUFInsts, FeatureFormattedMUBUFInsts,
- Feature32BitNumRecordsBufferResource,
+ Feature32BitNumRecordsBufferResource, FeatureGfx10BufferOOBModel,
FeatureMaxWavesPerEU16
]
>;
@@ -1888,7 +1900,7 @@ def FeatureGFX13 : GCNSubtargetFeatureGeneration<"GFX13",
FeatureAgentScopeFineGrainedRemoteMemoryAtomics, FeatureFlatOffsetBits24,
FeatureFlatSignedOffset, FeatureInstCacheLineSize128,
FeatureMTBUFInsts, FeatureFormattedMUBUFInsts,
- Feature45BitNumRecordsBufferResource,
+ Feature45BitNumRecordsBufferResource, FeatureGfx1250BufferOOBModel,
FeatureMaxWavesPerEU16
]
>;
@@ -2371,7 +2383,7 @@ def FeatureISAVersion11_7_Generic: FeatureSet<
def FeatureISAVersion12 : FeatureSet<
[FeatureGFX12,
- Feature32BitNumRecordsBufferResource,
+ Feature32BitNumRecordsBufferResource, FeatureGfx10BufferOOBModel,
FeatureSupportsWave64, FeatureSupportsWGP,
FeatureBackOffBarrier,
FeatureAddressableLocalMemorySize65536,
@@ -2520,6 +2532,7 @@ def FeatureISAVersion12_50_Common : FeatureSet<
FeatureSetPrioIncWgInst,
FeatureSWakeupBarrier,
Feature45BitNumRecordsBufferResource,
+ FeatureGfx1250BufferOOBModel,
FeatureClusters,
FeatureD16Writes32BitVgpr,
FeatureMcastLoadInsts,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
index 7721114fdd2f7..2f9299919d08c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
@@ -1175,6 +1175,79 @@ static Instruction *foldConstantIntoDotAccumulator(IntrinsicInst &II,
return IC.replaceOperand(II, AccIdx, NewAcc);
}
+// Check raw buffer access is OOB or not. For partial OOB access, just bail out.
+static bool isRawBufferAccessDefinitelyOOB(const GCNSubtarget *ST, Value *Rsrc,
+ Value *OffsetV, Value *SOffsetV,
+ Value *AuxV, uint64_t Payload,
+ const DataLayout &DL) {
+ using OOBModel = GCNSubtarget::AMDGPUBufferOOBModel;
+ std::optional<OOBModel> Model = ST->getBufferOOBModel();
+ std::optional<unsigned> NumRecordsWidth =
+ ST->getBufferResourceNumRecordsWidth();
+ if (!Model || !NumRecordsWidth)
+ return false;
+
+ auto *AuxC = dyn_cast<ConstantInt>(AuxV);
+ if (!AuxC)
+ return false;
+
+ uint64_t Offset = computeKnownBits(OffsetV, DL).getMinValue().getZExtValue();
+ uint64_t SOffset =
+ computeKnownBits(SOffsetV, DL).getMinValue().getZExtValue();
+
+ uint64_t Aux = AuxC->getZExtValue();
+ if (Aux & AMDGPU::CPol::VOLATILE)
+ return false;
+
+ // The resource descriptor must also be a compile-time constant.
+ const APInt *StrideAP, *NumRecAP, *FlagsAP;
+ if (!match(Rsrc, m_Intrinsic<Intrinsic::amdgcn_make_buffer_rsrc>(
+ m_Value(), m_APInt(StrideAP), m_APInt(NumRecAP),
+ m_APInt(FlagsAP))))
+ return false;
+
+ uint64_t StrideRaw = StrideAP->getZExtValue();
+ uint64_t Stride = StrideRaw & 0x3fff;
+ uint64_t Flags = FlagsAP->getZExtValue();
+ uint64_t NumRecordsMask = maskTrailingOnes<uint64_t>(*NumRecordsWidth);
+ uint64_t NumRecords =
+ NumRecAP->zextOrTrunc(64).getZExtValue() & NumRecordsMask;
+
+ switch (*Model) {
+ case OOBModel::Gfx9:
+ case OOBModel::Gfx10: {
+ bool AddTid = (Flags >> 23) & 1;
+ bool Swizzle = AMDGPU::isGFX11Plus(*ST) ? (StrideRaw >> 14) != 0
+ : (StrideRaw >> 15) != 0;
+ if (AddTid || Swizzle)
+ return false;
+
+ uint64_t Bound = NumRecords > SOffset ? NumRecords - SOffset : 0;
+ if (*Model == OOBModel::Gfx10) {
+ unsigned OOBSelect = (Flags >> 28) & 3;
+ return OOBSelect == 3 && Offset + Payload > Bound;
+ }
+ // Nonzero stride drops soffset out of the gfx9 mode equation.
+ return Offset + Payload > (Stride == 0 ? Bound : NumRecords);
+ }
+ case OOBModel::Gfx1250: {
+ bool IsBuffer = ((Flags >> 2) & 3) == 0;
+ bool Swizzle = Flags & 1;
+ bool StrideScaled = (StrideRaw >> 14) != 0;
+ bool OOBCheckDisabled = NumRecords == NumRecordsMask || NumRecords == 0;
+ if (!IsBuffer || Swizzle || StrideScaled || OOBCheckDisabled)
+ return false;
+
+ unsigned OOBSelect = (Flags >> 1) & 1;
+ // Here SOffset is added directly into the offset used for bound check.
+ uint64_t Addr = Offset + SOffset;
+ return Addr + Payload > NumRecords ||
+ (OOBSelect == 1 && Addr + Payload > Stride);
+ }
+ }
+ return false;
+}
+
std::optional<Instruction *>
GCNTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
Intrinsic::ID IID = II.getIntrinsicID();
@@ -2104,6 +2177,20 @@ GCNTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
return IC.replaceInstUsesWith(II, PoisonValue::get(II.getType()));
return std::nullopt;
}
+ case Intrinsic::amdgcn_raw_ptr_buffer_load: {
+ if (!isRawBufferAccessDefinitelyOOB(
+ ST, II.getArgOperand(0), II.getArgOperand(1), II.getArgOperand(2),
+ II.getArgOperand(3), /*Payload=*/1, IC.getDataLayout()))
+ break;
+ return IC.replaceInstUsesWith(II, Constant::getNullValue(II.getType()));
+ }
+ case Intrinsic::amdgcn_raw_ptr_buffer_store: {
+ if (!isRawBufferAccessDefinitelyOOB(
+ ST, II.getArgOperand(1), II.getArgOperand(2), II.getArgOperand(3),
+ II.getArgOperand(4), /*Payload=*/1, IC.getDataLayout()))
+ break;
+ return IC.eraseInstFromFunction(II);
+ }
case Intrinsic::amdgcn_raw_buffer_store_format:
case Intrinsic::amdgcn_struct_buffer_store_format:
case Intrinsic::amdgcn_raw_tbuffer_store:
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 2e09f1f1fda39..54d89462d9368 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -59,6 +59,12 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
LLVMAMDHSADebugTrap = 0x03,
};
+ enum class AMDGPUBufferOOBModel {
+ Gfx9 = 1,
+ Gfx10 = 2,
+ Gfx1250 = 3,
+ };
+
private:
/// SelectionDAGISel related APIs.
std::unique_ptr<const SelectionDAGTargetInfo> TSInfo;
@@ -88,6 +94,9 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
/// set from tablegen subtarget features, 0 is unknown.
unsigned BufferResourceNumRecordsWidth = 0;
+ /// OOB range-check model, set from tablegen subtarget features. 0 is unknown.
+ unsigned BufferOOBModel = 0;
+
// Dynamically set bits that enable features.
bool ScalarizeGlobal = false;
const bool BufferOOBRelaxed;
@@ -357,6 +366,14 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
return BufferResourceNumRecordsWidth;
}
+ /// Return the OOB range-check model this hardware implements or nullopt if
+ /// unknown.
+ std::optional<AMDGPUBufferOOBModel> getBufferOOBModel() const {
+ if (BufferOOBModel == 0)
+ return std::nullopt;
+ return static_cast<AMDGPUBufferOOBModel>(BufferOOBModel);
+ }
+
bool isCuModeEnabled() const { return EnableCuMode; }
/// \returns Whether a work-group runs on all of the block's SIMDs.
>From eda4560b5687b82b70b3064ace866deb5f22db8f Mon Sep 17 00:00:00 2001
From: Dark Steve Jobs <Prasoon.Mishra at amd.com>
Date: Tue, 29 Sep 2026 22:18:10 +0530
Subject: [PATCH 2/3] Add tests.
---
.../InstCombine/AMDGPU/fold-oob-raw-buffer.ll | 250 ++++--------------
1 file changed, 58 insertions(+), 192 deletions(-)
diff --git a/llvm/test/Transforms/InstCombine/AMDGPU/fold-oob-raw-buffer.ll b/llvm/test/Transforms/InstCombine/AMDGPU/fold-oob-raw-buffer.ll
index fd7050d4854fb..19fa48a02d81d 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/fold-oob-raw-buffer.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/fold-oob-raw-buffer.ll
@@ -12,9 +12,7 @@
define i32 @gfx9_stride_zero_soffset_subtracted_fold(ptr %p) {
; GFX9-LABEL: define i32 @gfx9_stride_zero_soffset_subtracted_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 8, i32 8, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @gfx9_stride_zero_soffset_subtracted_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
@@ -36,15 +34,11 @@ define i32 @gfx9_stride_zero_soffset_subtracted_fold(ptr %p) {
;
; GFX1250-LABEL: define i32 @gfx9_stride_zero_soffset_subtracted_fold(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX1250-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 8, i32 8, i32 0)
-; GFX1250-NEXT: ret i32 [[V]]
+; GFX1250-NEXT: ret i32 0
;
; GFX13-LABEL: define i32 @gfx9_stride_zero_soffset_subtracted_fold(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX13-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 8, i32 8, i32 0)
-; GFX13-NEXT: ret i32 [[V]]
+; GFX13-NEXT: ret i32 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 0)
%v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 8, i32 8, i32 0)
@@ -55,9 +49,7 @@ define i32 @gfx9_stride_zero_soffset_subtracted_fold(ptr %p) {
define i32 @gfx9_stride_nonzero_soffset_ignored_fold(ptr %p) {
; GFX9-LABEL: define i32 @gfx9_stride_nonzero_soffset_ignored_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 4, i64 16, i32 0)
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 16, i32 100, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @gfx9_stride_nonzero_soffset_ignored_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -79,15 +71,11 @@ define i32 @gfx9_stride_nonzero_soffset_ignored_fold(ptr %p) {
;
; GFX1250-LABEL: define i32 @gfx9_stride_nonzero_soffset_ignored_fold(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 4, i64 16, i32 0)
-; GFX1250-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 16, i32 100, i32 0)
-; GFX1250-NEXT: ret i32 [[V]]
+; GFX1250-NEXT: ret i32 0
;
; GFX13-LABEL: define i32 @gfx9_stride_nonzero_soffset_ignored_fold(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 4, i64 16, i32 0)
-; GFX13-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 16, i32 100, i32 0)
-; GFX13-NEXT: ret i32 [[V]]
+; GFX13-NEXT: ret i32 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 4, i64 16, i32 0)
%v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 16, i32 100, i32 0)
@@ -122,15 +110,11 @@ define i32 @gfx9_addtid_no_fold(ptr %p) {
;
; GFX1250-LABEL: define i32 @gfx9_addtid_no_fold(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 8388608)
-; GFX1250-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT: ret i32 [[V]]
+; GFX1250-NEXT: ret i32 0
;
; GFX13-LABEL: define i32 @gfx9_addtid_no_fold(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 8388608)
-; GFX13-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT: ret i32 [[V]]
+; GFX13-NEXT: ret i32 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 8388608)
%v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -227,9 +211,7 @@ define i32 @gfx9_volatile_no_fold(ptr %p) {
define i32 @gfx9_boundary_offset_equals_bound_fold(ptr %p) {
; GFX9-LABEL: define i32 @gfx9_boundary_offset_equals_bound_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 64, i32 0)
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 64, i32 0, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @gfx9_boundary_offset_equals_bound_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -251,15 +233,11 @@ define i32 @gfx9_boundary_offset_equals_bound_fold(ptr %p) {
;
; GFX1250-LABEL: define i32 @gfx9_boundary_offset_equals_bound_fold(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 64, i32 0)
-; GFX1250-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 64, i32 0, i32 0)
-; GFX1250-NEXT: ret i32 [[V]]
+; GFX1250-NEXT: ret i32 0
;
; GFX13-LABEL: define i32 @gfx9_boundary_offset_equals_bound_fold(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 64, i32 0)
-; GFX13-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 64, i32 0, i32 0)
-; GFX13-NEXT: ret i32 [[V]]
+; GFX13-NEXT: ret i32 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 64, i32 0)
%v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 64, i32 0, i32 0)
@@ -315,39 +293,27 @@ define i32 @gfx9_boundary_offset_below_bound_no_fold(ptr %p) {
define i32 @gfx10_oobselect3_fold(ptr %p) {
; GFX9-LABEL: define i32 @gfx10_oobselect3_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @gfx10_oobselect3_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX10-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX10-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX10-NEXT: ret i32 [[V]]
+; GFX10-NEXT: ret i32 0
;
; GFX11-LABEL: define i32 @gfx10_oobselect3_fold(
; GFX11-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX11-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX11-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX11-NEXT: ret i32 [[V]]
+; GFX11-NEXT: ret i32 0
;
; GFX12-LABEL: define i32 @gfx10_oobselect3_fold(
; GFX12-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX12-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX12-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX12-NEXT: ret i32 [[V]]
+; GFX12-NEXT: ret i32 0
;
; GFX1250-LABEL: define i32 @gfx10_oobselect3_fold(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX1250-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT: ret i32 [[V]]
+; GFX1250-NEXT: ret i32 0
;
; GFX13-LABEL: define i32 @gfx10_oobselect3_fold(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX13-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT: ret i32 [[V]]
+; GFX13-NEXT: ret i32 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 805306368)
%v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -358,9 +324,7 @@ define i32 @gfx10_oobselect3_fold(ptr %p) {
define i32 @gfx10_oobselect0_no_fold(ptr %p) {
; GFX9-LABEL: define i32 @gfx10_oobselect0_no_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @gfx10_oobselect0_no_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -382,15 +346,11 @@ define i32 @gfx10_oobselect0_no_fold(ptr %p) {
;
; GFX1250-LABEL: define i32 @gfx10_oobselect0_no_fold(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX1250-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT: ret i32 [[V]]
+; GFX1250-NEXT: ret i32 0
;
; GFX13-LABEL: define i32 @gfx10_oobselect0_no_fold(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX13-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT: ret i32 [[V]]
+; GFX13-NEXT: ret i32 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 0)
%v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -425,15 +385,11 @@ define i32 @gfx10_addtid_no_fold(ptr %p) {
;
; GFX1250-LABEL: define i32 @gfx10_addtid_no_fold(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 813694976)
-; GFX1250-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT: ret i32 [[V]]
+; GFX1250-NEXT: ret i32 0
;
; GFX13-LABEL: define i32 @gfx10_addtid_no_fold(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 813694976)
-; GFX13-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT: ret i32 [[V]]
+; GFX13-NEXT: ret i32 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 813694976)
%v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -444,39 +400,27 @@ define i32 @gfx10_addtid_no_fold(ptr %p) {
define i32 @gfx12_swizzle_disabled_fold(ptr %p) {
; GFX9-LABEL: define i32 @gfx12_swizzle_disabled_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @gfx12_swizzle_disabled_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX10-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX10-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX10-NEXT: ret i32 [[V]]
+; GFX10-NEXT: ret i32 0
;
; GFX11-LABEL: define i32 @gfx12_swizzle_disabled_fold(
; GFX11-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX11-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX11-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX11-NEXT: ret i32 [[V]]
+; GFX11-NEXT: ret i32 0
;
; GFX12-LABEL: define i32 @gfx12_swizzle_disabled_fold(
; GFX12-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX12-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX12-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX12-NEXT: ret i32 [[V]]
+; GFX12-NEXT: ret i32 0
;
; GFX1250-LABEL: define i32 @gfx12_swizzle_disabled_fold(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX1250-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT: ret i32 [[V]]
+; GFX1250-NEXT: ret i32 0
;
; GFX13-LABEL: define i32 @gfx12_swizzle_disabled_fold(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX13-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT: ret i32 [[V]]
+; GFX13-NEXT: ret i32 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 805306368)
%v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -489,9 +433,7 @@ define i32 @gfx12_swizzle_disabled_fold(ptr %p) {
define i32 @gfx1250_nonbuffer_type_no_fold(ptr %p) {
; GFX9-LABEL: define i32 @gfx1250_nonbuffer_type_no_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 4)
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @gfx1250_nonbuffer_type_no_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -533,9 +475,7 @@ define i32 @gfx1250_nonbuffer_type_no_fold(ptr %p) {
define i32 @gfx1250_stride_scaled_no_fold(ptr %p) {
; GFX9-LABEL: define i32 @gfx1250_stride_scaled_no_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 16384, i64 1000000, i32 0)
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 2000000000, i32 0, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @gfx1250_stride_scaled_no_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -576,9 +516,7 @@ define i32 @gfx1250_stride_scaled_no_fold(ptr %p) {
define i32 @gfx1250_numrecords_zero_no_fold(ptr %p) {
; GFX9-LABEL: define i32 @gfx1250_numrecords_zero_no_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 0, i32 0)
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 100, i32 0, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @gfx1250_numrecords_zero_no_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -620,9 +558,7 @@ define i32 @gfx1250_numrecords_zero_no_fold(ptr %p) {
define i32 @gfx1250_numrecords_allones_no_fold(ptr %p) {
; GFX9-LABEL: define i32 @gfx1250_numrecords_allones_no_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 35184372088831, i32 0)
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 -1, i32 0, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @gfx1250_numrecords_allones_no_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -663,9 +599,7 @@ define i32 @gfx1250_numrecords_allones_no_fold(ptr %p) {
define i32 @gfx1250_oobselect0_fold(ptr %p) {
; GFX9-LABEL: define i32 @gfx1250_oobselect0_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @gfx1250_oobselect0_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -687,15 +621,11 @@ define i32 @gfx1250_oobselect0_fold(ptr %p) {
;
; GFX1250-LABEL: define i32 @gfx1250_oobselect0_fold(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX1250-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT: ret i32 [[V]]
+; GFX1250-NEXT: ret i32 0
;
; GFX13-LABEL: define i32 @gfx1250_oobselect0_fold(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX13-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT: ret i32 [[V]]
+; GFX13-NEXT: ret i32 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 0)
%v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -730,15 +660,11 @@ define i32 @gfx1250_oobselect1_stride0_always_oob(ptr %p) {
;
; GFX1250-LABEL: define i32 @gfx1250_oobselect1_stride0_always_oob(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 1000000, i32 2)
-; GFX1250-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 0, i32 0, i32 0)
-; GFX1250-NEXT: ret i32 [[V]]
+; GFX1250-NEXT: ret i32 0
;
; GFX13-LABEL: define i32 @gfx1250_oobselect1_stride0_always_oob(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 1000000, i32 2)
-; GFX13-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 0, i32 0, i32 0)
-; GFX13-NEXT: ret i32 [[V]]
+; GFX13-NEXT: ret i32 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 1000000, i32 2)
%v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 0, i32 0, i32 0)
@@ -750,9 +676,7 @@ define i32 @gfx1250_oobselect1_stride0_always_oob(ptr %p) {
define i32 @gfx1250_soffset_alone_sufficient_fold(ptr %p) {
; GFX9-LABEL: define i32 @gfx1250_soffset_alone_sufficient_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 0, i32 32, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @gfx1250_soffset_alone_sufficient_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
@@ -774,15 +698,11 @@ define i32 @gfx1250_soffset_alone_sufficient_fold(ptr %p) {
;
; GFX1250-LABEL: define i32 @gfx1250_soffset_alone_sufficient_fold(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX1250-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 0, i32 32, i32 0)
-; GFX1250-NEXT: ret i32 [[V]]
+; GFX1250-NEXT: ret i32 0
;
; GFX13-LABEL: define i32 @gfx1250_soffset_alone_sufficient_fold(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 0)
-; GFX13-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 0, i32 32, i32 0)
-; GFX13-NEXT: ret i32 [[V]]
+; GFX13-NEXT: ret i32 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 0)
%v = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %rsrc, i32 0, i32 32, i32 0)
@@ -795,45 +715,27 @@ define i32 @gfx1250_soffset_alone_sufficient_fold(ptr %p) {
define i32 @known_bits_offset_provable_fold(ptr %p, i32 %x) {
; GFX9-LABEL: define i32 @known_bits_offset_provable_fold(
; GFX9-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX9-NEXT: [[OFF:%.*]] = or i32 [[X]], 2048
-; GFX9-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 [[OFF]], i32 0, i32 0)
-; GFX9-NEXT: ret i32 [[V]]
+; GFX9-NEXT: ret i32 0
;
; GFX10-LABEL: define i32 @known_bits_offset_provable_fold(
; GFX10-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) #[[ATTR0]] {
-; GFX10-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX10-NEXT: [[OFF:%.*]] = or i32 [[X]], 2048
-; GFX10-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 [[OFF]], i32 0, i32 0)
-; GFX10-NEXT: ret i32 [[V]]
+; GFX10-NEXT: ret i32 0
;
; GFX11-LABEL: define i32 @known_bits_offset_provable_fold(
; GFX11-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) #[[ATTR0]] {
-; GFX11-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX11-NEXT: [[OFF:%.*]] = or i32 [[X]], 2048
-; GFX11-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 [[OFF]], i32 0, i32 0)
-; GFX11-NEXT: ret i32 [[V]]
+; GFX11-NEXT: ret i32 0
;
; GFX12-LABEL: define i32 @known_bits_offset_provable_fold(
; GFX12-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) #[[ATTR0]] {
-; GFX12-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX12-NEXT: [[OFF:%.*]] = or i32 [[X]], 2048
-; GFX12-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 [[OFF]], i32 0, i32 0)
-; GFX12-NEXT: ret i32 [[V]]
+; GFX12-NEXT: ret i32 0
;
; GFX1250-LABEL: define i32 @known_bits_offset_provable_fold(
; GFX1250-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX1250-NEXT: [[OFF:%.*]] = or i32 [[X]], 2048
-; GFX1250-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 [[OFF]], i32 0, i32 0)
-; GFX1250-NEXT: ret i32 [[V]]
+; GFX1250-NEXT: ret i32 0
;
; GFX13-LABEL: define i32 @known_bits_offset_provable_fold(
; GFX13-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX13-NEXT: [[OFF:%.*]] = or i32 [[X]], 2048
-; GFX13-NEXT: [[V:%.*]] = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) [[RSRC]], i32 [[OFF]], i32 0, i32 0)
-; GFX13-NEXT: ret i32 [[V]]
+; GFX13-NEXT: ret i32 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 805306368)
%off = or i32 %x, 2048
@@ -940,39 +842,27 @@ define i32 @known_bits_neither_sufficient_no_fold(ptr %p, i32 %x, i32 %y) {
define <4 x i32> @wide_vector_load_fold(ptr %p) {
; GFX9-LABEL: define <4 x i32> @wide_vector_load_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX9-NEXT: [[V:%.*]] = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT: ret <4 x i32> [[V]]
+; GFX9-NEXT: ret <4 x i32> zeroinitializer
;
; GFX10-LABEL: define <4 x i32> @wide_vector_load_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX10-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX10-NEXT: [[V:%.*]] = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX10-NEXT: ret <4 x i32> [[V]]
+; GFX10-NEXT: ret <4 x i32> zeroinitializer
;
; GFX11-LABEL: define <4 x i32> @wide_vector_load_fold(
; GFX11-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX11-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX11-NEXT: [[V:%.*]] = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX11-NEXT: ret <4 x i32> [[V]]
+; GFX11-NEXT: ret <4 x i32> zeroinitializer
;
; GFX12-LABEL: define <4 x i32> @wide_vector_load_fold(
; GFX12-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX12-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX12-NEXT: [[V:%.*]] = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX12-NEXT: ret <4 x i32> [[V]]
+; GFX12-NEXT: ret <4 x i32> zeroinitializer
;
; GFX1250-LABEL: define <4 x i32> @wide_vector_load_fold(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX1250-NEXT: [[V:%.*]] = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT: ret <4 x i32> [[V]]
+; GFX1250-NEXT: ret <4 x i32> zeroinitializer
;
; GFX13-LABEL: define <4 x i32> @wide_vector_load_fold(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX13-NEXT: [[V:%.*]] = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT: ret <4 x i32> [[V]]
+; GFX13-NEXT: ret <4 x i32> zeroinitializer
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 805306368)
%v = call <4 x i32> @llvm.amdgcn.raw.ptr.buffer.load.v4i32(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -983,39 +873,27 @@ define <4 x i32> @wide_vector_load_fold(ptr %p) {
define i64 @wide_i64_load_fold(ptr %p) {
; GFX9-LABEL: define i64 @wide_i64_load_fold(
; GFX9-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX9-NEXT: [[V:%.*]] = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX9-NEXT: ret i64 [[V]]
+; GFX9-NEXT: ret i64 0
;
; GFX10-LABEL: define i64 @wide_i64_load_fold(
; GFX10-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX10-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX10-NEXT: [[V:%.*]] = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX10-NEXT: ret i64 [[V]]
+; GFX10-NEXT: ret i64 0
;
; GFX11-LABEL: define i64 @wide_i64_load_fold(
; GFX11-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX11-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX11-NEXT: [[V:%.*]] = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX11-NEXT: ret i64 [[V]]
+; GFX11-NEXT: ret i64 0
;
; GFX12-LABEL: define i64 @wide_i64_load_fold(
; GFX12-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX12-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX12-NEXT: [[V:%.*]] = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX12-NEXT: ret i64 [[V]]
+; GFX12-NEXT: ret i64 0
;
; GFX1250-LABEL: define i64 @wide_i64_load_fold(
; GFX1250-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX1250-NEXT: [[V:%.*]] = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX1250-NEXT: ret i64 [[V]]
+; GFX1250-NEXT: ret i64 0
;
; GFX13-LABEL: define i64 @wide_i64_load_fold(
; GFX13-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX13-NEXT: [[V:%.*]] = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
-; GFX13-NEXT: ret i64 [[V]]
+; GFX13-NEXT: ret i64 0
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 805306368)
%v = call i64 @llvm.amdgcn.raw.ptr.buffer.load.i64(ptr addrspace(8) %rsrc, i32 32, i32 0, i32 0)
@@ -1026,38 +904,26 @@ define i64 @wide_i64_load_fold(ptr %p) {
define void @wide_vector_store_erased(ptr %p, <4 x i32> %val) {
; GFX9-LABEL: define void @wide_vector_store_erased(
; GFX9-SAME: ptr [[P:%.*]], <4 x i32> [[VAL:%.*]]) #[[ATTR0]] {
-; GFX9-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX9-NEXT: call void @llvm.amdgcn.raw.ptr.buffer.store.v4i32(<4 x i32> [[VAL]], ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
; GFX9-NEXT: ret void
;
; GFX10-LABEL: define void @wide_vector_store_erased(
; GFX10-SAME: ptr [[P:%.*]], <4 x i32> [[VAL:%.*]]) #[[ATTR0]] {
-; GFX10-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX10-NEXT: call void @llvm.amdgcn.raw.ptr.buffer.store.v4i32(<4 x i32> [[VAL]], ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
; GFX10-NEXT: ret void
;
; GFX11-LABEL: define void @wide_vector_store_erased(
; GFX11-SAME: ptr [[P:%.*]], <4 x i32> [[VAL:%.*]]) #[[ATTR0]] {
-; GFX11-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX11-NEXT: call void @llvm.amdgcn.raw.ptr.buffer.store.v4i32(<4 x i32> [[VAL]], ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
; GFX11-NEXT: ret void
;
; GFX12-LABEL: define void @wide_vector_store_erased(
; GFX12-SAME: ptr [[P:%.*]], <4 x i32> [[VAL:%.*]]) #[[ATTR0]] {
-; GFX12-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX12-NEXT: call void @llvm.amdgcn.raw.ptr.buffer.store.v4i32(<4 x i32> [[VAL]], ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
; GFX12-NEXT: ret void
;
; GFX1250-LABEL: define void @wide_vector_store_erased(
; GFX1250-SAME: ptr [[P:%.*]], <4 x i32> [[VAL:%.*]]) #[[ATTR0]] {
-; GFX1250-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX1250-NEXT: call void @llvm.amdgcn.raw.ptr.buffer.store.v4i32(<4 x i32> [[VAL]], ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
; GFX1250-NEXT: ret void
;
; GFX13-LABEL: define void @wide_vector_store_erased(
; GFX13-SAME: ptr [[P:%.*]], <4 x i32> [[VAL:%.*]]) #[[ATTR0]] {
-; GFX13-NEXT: [[RSRC:%.*]] = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr [[P]], i16 0, i64 16, i32 805306368)
-; GFX13-NEXT: call void @llvm.amdgcn.raw.ptr.buffer.store.v4i32(<4 x i32> [[VAL]], ptr addrspace(8) [[RSRC]], i32 32, i32 0, i32 0)
; GFX13-NEXT: ret void
;
%rsrc = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p0.i64(ptr %p, i16 0, i64 16, i32 805306368)
>From 418d68df1c3b52d0cf30f3f83ebd2830279f4272 Mon Sep 17 00:00:00 2001
From: Dark Steve Jobs <Prasoon.Mishra at amd.com>
Date: Thu, 1 Oct 2026 14:18:39 +0530
Subject: [PATCH 3/3] Addressed the comments.
---
llvm/lib/Target/AMDGPU/AMDGPU.td | 5 ++++-
llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp | 1 +
2 files changed, 5 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index 28dc132559828..3487e08371890 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -1612,7 +1612,10 @@ class FeatureNumRecordsBufferResource<int width> : SubtargetFeature<
def Feature32BitNumRecordsBufferResource : FeatureNumRecordsBufferResource<32>;
def Feature45BitNumRecordsBufferResource : FeatureNumRecordsBufferResource<45>;
-// The out-of-bounds range-check model.
+// The out-of-bounds range-check model. The buffer resource (V#) has no
+// OOB_SELECT field on Gfx9, a 2-bit OOB_SELECT on Gfx10 (gfx10-gfx12),
+// and a 1-bit OOB_SELECT on Gfx1250 (gfx1250+). The OOB range-check
+// equations differ per model.
class FeatureBufferOOBModel<string model, int value> : SubtargetFeature<
model#"-buffer-oob-model",
"BufferOOBModel",
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
index 2f9299919d08c..5e3c2242a3311 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
@@ -1222,6 +1222,7 @@ static bool isRawBufferAccessDefinitelyOOB(const GCNSubtarget *ST, Value *Rsrc,
if (AddTid || Swizzle)
return false;
+ // Saturating uint64 subtract: clamp at 0 if SOffset >= NumRecords.
uint64_t Bound = NumRecords > SOffset ? NumRecords - SOffset : 0;
if (*Model == OOBModel::Gfx10) {
unsigned OOBSelect = (Flags >> 28) & 3;
More information about the llvm-branch-commits
mailing list