[llvm] [AMDGPU] Split large raw buffer offsets to SOFFSET where allowed (PR #207821)
Matt Arsenault via llvm-commits
llvm-commits at lists.llvm.org
Thu Jul 9 12:55:58 PDT 2026
================
@@ -6493,60 +6494,115 @@ bool AMDGPULegalizerInfo::legalizeIsAddrSpace(MachineInstr &MI,
return true;
}
+static bool canMoveBufferOffsetToSOffset(const GCNSubtarget &ST,
+ bool HasLinearOOB, bool IsNUW,
+ uint32_t Offset) {
+ // There is a hardware bug in SI and CI which prevents address clamping in
+ // MUBUF instructions from working correctly with SOffsets. The immediate
+ // offset is unaffected.
+ return HasLinearOOB && IsNUW && static_cast<int32_t>(Offset) > 0 &&
+ ST.getGeneration() > AMDGPUSubtarget::SEA_ISLANDS;
+}
+
+static std::optional<uint32_t>
+getCombinedBufferSOffset(MachineRegisterInfo &MRI, Register SOffset,
+ uint32_t Offset) {
+ APInt C;
+ if (!mi_match(SOffset, MRI, m_ICst(C)))
+ return std::nullopt;
+ return checkedAddUnsigned<uint32_t>(static_cast<uint32_t>(C.getZExtValue()),
+ Offset);
+}
+
// The raw.(t)buffer and struct.(t)buffer intrinsics have two offset args:
// offset (the offset that is included in bounds checking and swizzling, to be
// split between the instruction's voffset and immoffset fields) and soffset
// (the offset that is excluded from bounds checking and swizzling, to go in
-// the instruction's soffset field). This function takes the first kind of
-// offset and figures out how to split it between voffset and immoffset.
-std::pair<Register, unsigned>
+// the instruction's soffset field, except that it does affect num_records and
+// so can (if there's no unsigned overflow) be used for bounds checking in raw.*
+// intrinsics). This function takes both offsets and figures out how to split
+// them between voffset, soffset, and immoffset.
+std::tuple<Register, Register, unsigned>
AMDGPULegalizerInfo::splitBufferOffsets(MachineIRBuilder &B,
- Register OrigOffset) const {
+ Register OrigOffset,
+ Register OrigSOffset, bool HasLinearOOB,
+ GISelValueTracking *VT) const {
const unsigned MaxImm = SIInstrInfo::getMaxMUBUFImmOffset(ST);
Register BaseReg;
unsigned ImmOffset;
const LLT S32 = LLT::scalar(32);
MachineRegisterInfo &MRI = *B.getMRI();
-
+ Register SOffset = OrigSOffset;
// On GFX1250+, voffset and immoffset are zero-extended from 32 bits before
// being added, so we can only safely match a 32-bit addition with no unsigned
// overflow.
bool CheckNUW = ST.hasGFX1250Insts();
- std::tie(BaseReg, ImmOffset) = AMDGPU::getBaseWithConstantOffset(
- MRI, OrigOffset, /*KnownBits=*/nullptr, CheckNUW);
+ std::tie(BaseReg, ImmOffset) =
+ AMDGPU::getBaseWithConstantOffset(MRI, OrigOffset, VT, CheckNUW);
+ bool IsNUW = false;
+ if (ImmOffset) {
+ if (!BaseReg) {
+ IsNUW = true;
+ } else if (MachineInstr *OffsetDef =
+ getDefIgnoringCopies(OrigOffset, MRI)) {
+ // Match SelectionDAG: a G_OR with a constant reaches here only when it is
+ // add-like, and such an OR satisfies the no-wrap condition.
+ IsNUW = OffsetDef->getOpcode() == TargetOpcode::G_OR ||
+ (OffsetDef->getOpcode() == TargetOpcode::G_ADD &&
+ OffsetDef->getFlag(MachineInstr::NoUWrap));
----------------
arsenm wrote:
There should be a matcher for this somewhere
https://github.com/llvm/llvm-project/pull/207821
More information about the llvm-commits
mailing list