[llvm] [amdgpu] Add HasBufferInv AMDGPUSubtargetFeature (PR #226254)

Dan Zimmerman via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 11:04:43 PDT 2026


https://github.com/danzimm created https://github.com/llvm/llvm-project/pull/226254

As suggested by @arsenm I've created a new `AMDGPUSubtargetFeature` to test whether `buffer_inv` is available, as opposed to just using `hasGFX940Insts` to gate whether or not it's available.

The aforementioned suggestion was in response to https://github.com/llvm/llvm-project/pull/221115. I plan to use this flag in that PR.

In addition to introducing this new feature, I use it in place of `hasGFX940Insts` in `SIMemoryLegalizer` where it clearly is checking whether it's safe to use `buffer_inv`. I didn't find any other obvious place in the codebase which uses `hasGFX940Insts` as a predicate for `buffer_inv`.

This was co-authored with codex. Every line was audited by me, though.

>From 26526a4a2392eae47dcc9e52ff9239823fbfec71 Mon Sep 17 00:00:00 2001
From: Dan Zimmerman <danzimm at meta.com>
Date: Wed, 23 Sep 2026 09:12:41 -0700
Subject: [PATCH 1/3] [AMDGPU] Add buffer_inv subtarget feature

---
 llvm/lib/Target/AMDGPU/AMDGPU.td                 |  8 +++++++-
 llvm/unittests/TargetParser/TargetParserTest.cpp | 12 ++++++++++++
 2 files changed, 19 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index c3b4d53a7effa..9f4a736ad5059 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -510,6 +510,10 @@ defm GFX940Insts : AMDGPUSubtargetFeature<"gfx940-insts",
   /*GenPredicate=*/0
 >;
 
+defm BufferInvInst : AMDGPUSubtargetFeature<"buffer-inv-inst",
+  "Has buffer_inv instruction"
+>;
+
 defm Permlane16Insts : AMDGPUSubtargetFeature<"permlane16-insts",
   "Has v_permlane16_b32/v_permlanex16_b32 instructions"
 >;
@@ -2067,6 +2071,7 @@ def FeatureISAVersion9_4_Common : FeatureSet<
    FeatureAGPRAlloc,
    FeatureTgSplitSupport,
    FeatureGFX940Insts,
+   FeatureBufferInvInst,
    FeatureRequiresAlignedVGPRs,
    FeatureFmaMixInsts,
    FeatureDLInsts,
@@ -3305,7 +3310,8 @@ def AMDGPUFrontendVisibleFeatures {
   FeatureAtomicFaddRtnInsts, FeatureAtomicFlatPkAdd16Insts, FeatureAtomicGlobalPkAddBF16Inst,
   FeatureBF16ConversionInsts,
   FeatureBF16PackedInsts, FeatureBF16TransInsts, FeatureBF8ConversionScaleInsts,
-  FeatureBVHRayTracingInsts, FeatureBitOp3Insts, FeatureCIInsts,
+  FeatureBVHRayTracingInsts, FeatureBitOp3Insts, FeatureBufferInvInst,
+  FeatureCIInsts,
   FeatureClusters, FeatureCubeInsts, FeatureCvtPkNormVOP2Insts,
   FeatureCvtSrPkBF16F32Inst, FeatureDLInsts, FeatureDPP,
   FeatureDot10Insts, FeatureDot11Insts, FeatureDot12Insts,
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 62d6f3a4c955b..2847942227df3 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -2875,6 +2875,18 @@ TEST(TargetParserTest, testAMDGPUgetFeatureBitset) {
   EXPECT_TRUE(Empty.empty());
 }
 
+TEST(TargetParserTest, testAMDGPUBufferInvInstFeature) {
+  auto Has = [](AMDGPU::GPUKind AK) {
+    return AMDGPU::getFeatureBitset(AK).test(AMDGPU::FEAT_BUFFER_INV_INST);
+  };
+
+  EXPECT_FALSE(Has(AMDGPU::GK_GFX90A));
+  EXPECT_TRUE(Has(AMDGPU::GK_GFX942));
+  EXPECT_TRUE(Has(AMDGPU::GK_GFX950));
+  EXPECT_TRUE(Has(AMDGPU::GK_GFX9_4_GENERIC));
+  EXPECT_FALSE(Has(AMDGPU::GK_GFX1250));
+}
+
 TEST(TargetParserTest, testAMDGPUHalfAddressableLDSFeature) {
   auto Has = [](AMDGPU::GPUKind AK) {
     return AMDGPU::getFeatureBitset(AK).test(

>From e4eb788cf1a1ec0a28e72e4a63dc544d4f4811b9 Mon Sep 17 00:00:00 2001
From: Dan Zimmerman <danzimm at meta.com>
Date: Thu, 24 Sep 2026 10:20:47 -0700
Subject: [PATCH 2/3] [AMDGPU] Use BufferInvInst for buffer_inv predicates

---
 llvm/lib/Target/AMDGPU/BUFInstructions.td          |  8 +++++---
 llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s | 10 ++++++++++
 2 files changed, 15 insertions(+), 3 deletions(-)
 create mode 100644 llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s

diff --git a/llvm/lib/Target/AMDGPU/BUFInstructions.td b/llvm/lib/Target/AMDGPU/BUFInstructions.td
index 057c0f103ce80..2b5d20c6849f7 100644
--- a/llvm/lib/Target/AMDGPU/BUFInstructions.td
+++ b/llvm/lib/Target/AMDGPU/BUFInstructions.td
@@ -1429,7 +1429,7 @@ let SubtargetPredicate = HasAtomicFMinFMaxF64GlobalInsts in {
 }
 
 def BUFFER_INV : MUBUF_Invalidate<"buffer_inv"> {
-  let SubtargetPredicate = isGFX940Plus;
+  let SubtargetPredicate = HasBufferInvInst;
   let has_glc = 1;
   let has_sccb = 1;
   let InOperandList = (ins CPol_0:$cpol);
@@ -3660,10 +3660,12 @@ let AsmString = BUFFER_WBL2.Mnemonic, // drop flags
 defm BUFFER_WBL2  : MUBUF_Real_gfx90a<0x28>;
 defm BUFFER_INVL2 : MUBUF_Real_gfx90a<0x29>;
 
-let SubtargetPredicate = isGFX940Plus in {
+let SubtargetPredicate = isGFX940Plus in
 def BUFFER_WBL2_gfx940  : MUBUF_Real_gfx940<0x28, BUFFER_WBL2>;
+
+let SubtargetPredicate = HasBufferInvInst,
+    AssemblerPredicate = HasBufferInvInst in
 def BUFFER_INV_gfx940   : MUBUF_Real_gfx940<0x29, BUFFER_INV>;
-}
 
 class MTBUF_Real_Base_vi <bits<4> op, MTBUF_Pseudo ps, int Enc> :
   MTBUF_Real<ps>,
diff --git a/llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s b/llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s
new file mode 100644
index 0000000000000..52d7c4e5eed49
--- /dev/null
+++ b/llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s
@@ -0,0 +1,10 @@
+// RUN: llvm-mc -triple=amdgpu9.42 -show-encoding %s | FileCheck --check-prefix=SUPPORTED %s
+// RUN: llvm-mc -triple=amdgpu9.50 -show-encoding %s | FileCheck --check-prefix=SUPPORTED %s
+// RUN: llvm-mc -triple=amdgcn-amd-amdhsa -mcpu=gfx9-4-generic -show-encoding %s | FileCheck --check-prefix=SUPPORTED %s
+// RUN: not llvm-mc -triple=amdgpu9.0a -filetype=null %s 2>&1 | FileCheck --check-prefix=GFX90A %s
+// RUN: not llvm-mc -triple=amdgpu9.42 -mattr=-buffer-inv-inst -filetype=null %s 2>&1 | FileCheck --check-prefix=DISABLED %s
+
+buffer_inv sc0 sc1
+// SUPPORTED: buffer_inv sc0 sc1                    ; encoding: [0x00,0xc0,0xa4,0xe0,0x00,0x00,0x00,0x00]
+// GFX90A: error: instruction not supported on this GPU (gfx90a): buffer_inv
+// DISABLED: error: instruction not supported on this GPU (gfx942): buffer_inv

>From c502d9ef51b0abd8129033f6ee6e174bdaff0090 Mon Sep 17 00:00:00 2001
From: Dan Zimmerman <danzimm at meta.com>
Date: Thu, 24 Sep 2026 10:21:38 -0700
Subject: [PATCH 3/3] [AMDGPU] Use BufferInvInst in memory legalizer

---
 llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp | 6 +++---
 1 file changed, 3 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp b/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
index e818e06e3bb04..10804d65b82b0 100644
--- a/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
+++ b/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
@@ -1396,7 +1396,7 @@ bool SIGfx6CacheControl::insertAcquire(MachineBasicBlock::iterator &MI,
   if (canAffectGlobalAddrSpace(AddrSpace)) {
     switch (Scope) {
     case SIAtomicScope::SYSTEM:
-      if (ST.hasGFX940Insts()) {
+      if (ST.hasBufferInvInst()) {
         // Ensures that following loads will not see stale remote VMEM data or
         // stale local VMEM data with MTYPE NC. Local VMEM data with MTYPE RW
         // and CC will never be stale due to the local memory probes.
@@ -1428,7 +1428,7 @@ bool SIGfx6CacheControl::insertAcquire(MachineBasicBlock::iterator &MI,
       }
       [[fallthrough]];
     case SIAtomicScope::AGENT:
-      if (ST.hasGFX940Insts()) {
+      if (ST.hasBufferInvInst()) {
         // Ensures that following loads will not see stale remote date or local
         // MTYPE NC global data. Local MTYPE RW and CC memory will never be
         // stale due to the memory probes.
@@ -1445,7 +1445,7 @@ bool SIGfx6CacheControl::insertAcquire(MachineBasicBlock::iterator &MI,
       break;
     case SIAtomicScope::WORKGROUP:
       if (TgSplitEnabled) {
-        if (ST.hasGFX940Insts()) {
+        if (ST.hasBufferInvInst()) {
           // In threadgroup split mode the waves of a work-group can be
           // executing on different CUs. Therefore need to invalidate the L1
           // which is per CU. Otherwise in non-threadgroup split mode all waves



More information about the llvm-commits mailing list