[llvm] [AMDGPU] Split large raw buffer offsets to SOFFSET where allowed (PR #207821)

Matt Arsenault via llvm-commits llvm-commits at lists.llvm.org
Thu Jul 9 12:55:58 PDT 2026


================
@@ -6493,60 +6494,115 @@ bool AMDGPULegalizerInfo::legalizeIsAddrSpace(MachineInstr &MI,
   return true;
 }
 
+static bool canMoveBufferOffsetToSOffset(const GCNSubtarget &ST,
+                                         bool HasLinearOOB, bool IsNUW,
+                                         uint32_t Offset) {
+  // There is a hardware bug in SI and CI which prevents address clamping in
+  // MUBUF instructions from working correctly with SOffsets. The immediate
+  // offset is unaffected.
+  return HasLinearOOB && IsNUW && static_cast<int32_t>(Offset) > 0 &&
+         ST.getGeneration() > AMDGPUSubtarget::SEA_ISLANDS;
+}
+
+static std::optional<uint32_t>
+getCombinedBufferSOffset(MachineRegisterInfo &MRI, Register SOffset,
+                         uint32_t Offset) {
+  APInt C;
+  if (!mi_match(SOffset, MRI, m_ICst(C)))
+    return std::nullopt;
+  return checkedAddUnsigned<uint32_t>(static_cast<uint32_t>(C.getZExtValue()),
+                                      Offset);
+}
+
 // The raw.(t)buffer and struct.(t)buffer intrinsics have two offset args:
 // offset (the offset that is included in bounds checking and swizzling, to be
 // split between the instruction's voffset and immoffset fields) and soffset
 // (the offset that is excluded from bounds checking and swizzling, to go in
-// the instruction's soffset field).  This function takes the first kind of
-// offset and figures out how to split it between voffset and immoffset.
-std::pair<Register, unsigned>
+// the instruction's soffset field, except that it does affect num_records and
+// so can (if there's no unsigned overflow) be used for bounds checking in raw.*
+// intrinsics). This function takes both offsets and figures out how to split
+// them between voffset, soffset, and immoffset.
+std::tuple<Register, Register, unsigned>
 AMDGPULegalizerInfo::splitBufferOffsets(MachineIRBuilder &B,
-                                        Register OrigOffset) const {
+                                        Register OrigOffset,
+                                        Register OrigSOffset, bool HasLinearOOB,
+                                        GISelValueTracking *VT) const {
   const unsigned MaxImm = SIInstrInfo::getMaxMUBUFImmOffset(ST);
   Register BaseReg;
   unsigned ImmOffset;
   const LLT S32 = LLT::scalar(32);
   MachineRegisterInfo &MRI = *B.getMRI();
-
+  Register SOffset = OrigSOffset;
   // On GFX1250+, voffset and immoffset are zero-extended from 32 bits before
   // being added, so we can only safely match a 32-bit addition with no unsigned
   // overflow.
   bool CheckNUW = ST.hasGFX1250Insts();
-  std::tie(BaseReg, ImmOffset) = AMDGPU::getBaseWithConstantOffset(
-      MRI, OrigOffset, /*KnownBits=*/nullptr, CheckNUW);
+  std::tie(BaseReg, ImmOffset) =
+      AMDGPU::getBaseWithConstantOffset(MRI, OrigOffset, VT, CheckNUW);
+  bool IsNUW = false;
+  if (ImmOffset) {
+    if (!BaseReg) {
+      IsNUW = true;
+    } else if (MachineInstr *OffsetDef =
+                   getDefIgnoringCopies(OrigOffset, MRI)) {
+      // Match SelectionDAG: a G_OR with a constant reaches here only when it is
+      // add-like, and such an OR satisfies the no-wrap condition.
+      IsNUW = OffsetDef->getOpcode() == TargetOpcode::G_OR ||
+              (OffsetDef->getOpcode() == TargetOpcode::G_ADD &&
+               OffsetDef->getFlag(MachineInstr::NoUWrap));
----------------
arsenm wrote:

There should be a matcher for this somewhere 

https://github.com/llvm/llvm-project/pull/207821


More information about the llvm-commits mailing list