[llvm] [amdgpu] Add HasBufferInv AMDGPUSubtargetFeature (PR #226254)

Dan Zimmerman via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 14:48:40 PDT 2026


https://github.com/danzimm updated https://github.com/llvm/llvm-project/pull/226254

>From 57fb45d7ca23c516291ae89500092506e801aff4 Mon Sep 17 00:00:00 2001
From: Dan Zimmerman <danzimm at meta.com>
Date: Wed, 23 Sep 2026 09:12:41 -0700
Subject: [PATCH 1/5] [AMDGPU] Add buffer_inv subtarget feature

---
 llvm/lib/Target/AMDGPU/AMDGPU.td                 |  8 +++++++-
 llvm/unittests/TargetParser/TargetParserTest.cpp | 12 ++++++++++++
 2 files changed, 19 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index bab8ade0acc74c..f3011b5a1568f2 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -510,6 +510,10 @@ defm GFX940Insts : AMDGPUSubtargetFeature<"gfx940-insts",
   /*GenPredicate=*/0
 >;
 
+defm BufferInvInst : AMDGPUSubtargetFeature<"buffer-inv-inst",
+  "Has buffer_inv instruction"
+>;
+
 defm Permlane16Insts : AMDGPUSubtargetFeature<"permlane16-insts",
   "Has v_permlane16_b32/v_permlanex16_b32 instructions"
 >;
@@ -2073,6 +2077,7 @@ def FeatureISAVersion9_4_Common : FeatureSet<
    FeatureAGPRAlloc,
    FeatureTgSplitSupport,
    FeatureGFX940Insts,
+   FeatureBufferInvInst,
    FeatureRequiresAlignedVGPRs,
    FeatureFmaMixInsts,
    FeatureDLInsts,
@@ -3314,7 +3319,8 @@ def AMDGPUFrontendVisibleFeatures {
   FeatureAtomicFaddRtnInsts, FeatureAtomicFlatPkAdd16Insts, FeatureAtomicGlobalPkAddBF16Inst,
   FeatureBF16ConversionInsts,
   FeatureBF16PackedInsts, FeatureBF16TransInsts, FeatureBF8ConversionScaleInsts,
-  FeatureBVHRayTracingInsts, FeatureBitOp3Insts, FeatureCIInsts,
+  FeatureBVHRayTracingInsts, FeatureBitOp3Insts, FeatureBufferInvInst,
+  FeatureCIInsts,
   FeatureClusters, FeatureCubeInsts, FeatureCvtPkNormVOP2Insts,
   FeatureCvtSrPkBF16F32Inst, FeatureDLInsts, FeatureDPP,
   FeatureDot10Insts, FeatureDot11Insts, FeatureDot12Insts,
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 62d6f3a4c955b4..2847942227df33 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -2875,6 +2875,18 @@ TEST(TargetParserTest, testAMDGPUgetFeatureBitset) {
   EXPECT_TRUE(Empty.empty());
 }
 
+TEST(TargetParserTest, testAMDGPUBufferInvInstFeature) {
+  auto Has = [](AMDGPU::GPUKind AK) {
+    return AMDGPU::getFeatureBitset(AK).test(AMDGPU::FEAT_BUFFER_INV_INST);
+  };
+
+  EXPECT_FALSE(Has(AMDGPU::GK_GFX90A));
+  EXPECT_TRUE(Has(AMDGPU::GK_GFX942));
+  EXPECT_TRUE(Has(AMDGPU::GK_GFX950));
+  EXPECT_TRUE(Has(AMDGPU::GK_GFX9_4_GENERIC));
+  EXPECT_FALSE(Has(AMDGPU::GK_GFX1250));
+}
+
 TEST(TargetParserTest, testAMDGPUHalfAddressableLDSFeature) {
   auto Has = [](AMDGPU::GPUKind AK) {
     return AMDGPU::getFeatureBitset(AK).test(

>From 7e39f17f1cc326879bcb887d09538c5c0cba4398 Mon Sep 17 00:00:00 2001
From: Dan Zimmerman <danzimm at meta.com>
Date: Thu, 24 Sep 2026 10:20:47 -0700
Subject: [PATCH 2/5] [AMDGPU] Use BufferInvInst for buffer_inv predicates

---
 llvm/lib/Target/AMDGPU/BUFInstructions.td          |  8 +++++---
 llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s | 10 ++++++++++
 2 files changed, 15 insertions(+), 3 deletions(-)
 create mode 100644 llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s

diff --git a/llvm/lib/Target/AMDGPU/BUFInstructions.td b/llvm/lib/Target/AMDGPU/BUFInstructions.td
index 057c0f103ce800..2b5d20c6849f78 100644
--- a/llvm/lib/Target/AMDGPU/BUFInstructions.td
+++ b/llvm/lib/Target/AMDGPU/BUFInstructions.td
@@ -1429,7 +1429,7 @@ let SubtargetPredicate = HasAtomicFMinFMaxF64GlobalInsts in {
 }
 
 def BUFFER_INV : MUBUF_Invalidate<"buffer_inv"> {
-  let SubtargetPredicate = isGFX940Plus;
+  let SubtargetPredicate = HasBufferInvInst;
   let has_glc = 1;
   let has_sccb = 1;
   let InOperandList = (ins CPol_0:$cpol);
@@ -3660,10 +3660,12 @@ let AsmString = BUFFER_WBL2.Mnemonic, // drop flags
 defm BUFFER_WBL2  : MUBUF_Real_gfx90a<0x28>;
 defm BUFFER_INVL2 : MUBUF_Real_gfx90a<0x29>;
 
-let SubtargetPredicate = isGFX940Plus in {
+let SubtargetPredicate = isGFX940Plus in
 def BUFFER_WBL2_gfx940  : MUBUF_Real_gfx940<0x28, BUFFER_WBL2>;
+
+let SubtargetPredicate = HasBufferInvInst,
+    AssemblerPredicate = HasBufferInvInst in
 def BUFFER_INV_gfx940   : MUBUF_Real_gfx940<0x29, BUFFER_INV>;
-}
 
 class MTBUF_Real_Base_vi <bits<4> op, MTBUF_Pseudo ps, int Enc> :
   MTBUF_Real<ps>,
diff --git a/llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s b/llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s
new file mode 100644
index 00000000000000..52d7c4e5eed493
--- /dev/null
+++ b/llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s
@@ -0,0 +1,10 @@
+// RUN: llvm-mc -triple=amdgpu9.42 -show-encoding %s | FileCheck --check-prefix=SUPPORTED %s
+// RUN: llvm-mc -triple=amdgpu9.50 -show-encoding %s | FileCheck --check-prefix=SUPPORTED %s
+// RUN: llvm-mc -triple=amdgcn-amd-amdhsa -mcpu=gfx9-4-generic -show-encoding %s | FileCheck --check-prefix=SUPPORTED %s
+// RUN: not llvm-mc -triple=amdgpu9.0a -filetype=null %s 2>&1 | FileCheck --check-prefix=GFX90A %s
+// RUN: not llvm-mc -triple=amdgpu9.42 -mattr=-buffer-inv-inst -filetype=null %s 2>&1 | FileCheck --check-prefix=DISABLED %s
+
+buffer_inv sc0 sc1
+// SUPPORTED: buffer_inv sc0 sc1                    ; encoding: [0x00,0xc0,0xa4,0xe0,0x00,0x00,0x00,0x00]
+// GFX90A: error: instruction not supported on this GPU (gfx90a): buffer_inv
+// DISABLED: error: instruction not supported on this GPU (gfx942): buffer_inv

>From c2e5a971e71063816934e02bb9e47aeefe08524f Mon Sep 17 00:00:00 2001
From: Dan Zimmerman <danzimm at meta.com>
Date: Thu, 24 Sep 2026 10:21:38 -0700
Subject: [PATCH 3/5] [AMDGPU] Use BufferInvInst in memory legalizer

---
 llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp | 6 +++---
 1 file changed, 3 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp b/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
index e818e06e3bb048..10804d65b82b0f 100644
--- a/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
+++ b/llvm/lib/Target/AMDGPU/SIMemoryLegalizer.cpp
@@ -1396,7 +1396,7 @@ bool SIGfx6CacheControl::insertAcquire(MachineBasicBlock::iterator &MI,
   if (canAffectGlobalAddrSpace(AddrSpace)) {
     switch (Scope) {
     case SIAtomicScope::SYSTEM:
-      if (ST.hasGFX940Insts()) {
+      if (ST.hasBufferInvInst()) {
         // Ensures that following loads will not see stale remote VMEM data or
         // stale local VMEM data with MTYPE NC. Local VMEM data with MTYPE RW
         // and CC will never be stale due to the local memory probes.
@@ -1428,7 +1428,7 @@ bool SIGfx6CacheControl::insertAcquire(MachineBasicBlock::iterator &MI,
       }
       [[fallthrough]];
     case SIAtomicScope::AGENT:
-      if (ST.hasGFX940Insts()) {
+      if (ST.hasBufferInvInst()) {
         // Ensures that following loads will not see stale remote date or local
         // MTYPE NC global data. Local MTYPE RW and CC memory will never be
         // stale due to the memory probes.
@@ -1445,7 +1445,7 @@ bool SIGfx6CacheControl::insertAcquire(MachineBasicBlock::iterator &MI,
       break;
     case SIAtomicScope::WORKGROUP:
       if (TgSplitEnabled) {
-        if (ST.hasGFX940Insts()) {
+        if (ST.hasBufferInvInst()) {
           // In threadgroup split mode the waves of a work-group can be
           // executing on different CUs. Therefore need to invalidate the L1
           // which is per CU. Otherwise in non-threadgroup split mode all waves

>From e98f258eb828f03b16f047b6684c9090ff71cd46 Mon Sep 17 00:00:00 2001
From: Dan Zimmerman <danzimm at meta.com>
Date: Thu, 24 Sep 2026 14:44:56 -0700
Subject: [PATCH 4/5] [AMDGPU] Remove redundant buffer_inv feature tests

---
 llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s | 10 ----------
 llvm/unittests/TargetParser/TargetParserTest.cpp   | 12 ------------
 2 files changed, 22 deletions(-)
 delete mode 100644 llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s

diff --git a/llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s b/llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s
deleted file mode 100644
index 52d7c4e5eed493..00000000000000
--- a/llvm/test/MC/AMDGPU/buffer-inv-subtarget-feature.s
+++ /dev/null
@@ -1,10 +0,0 @@
-// RUN: llvm-mc -triple=amdgpu9.42 -show-encoding %s | FileCheck --check-prefix=SUPPORTED %s
-// RUN: llvm-mc -triple=amdgpu9.50 -show-encoding %s | FileCheck --check-prefix=SUPPORTED %s
-// RUN: llvm-mc -triple=amdgcn-amd-amdhsa -mcpu=gfx9-4-generic -show-encoding %s | FileCheck --check-prefix=SUPPORTED %s
-// RUN: not llvm-mc -triple=amdgpu9.0a -filetype=null %s 2>&1 | FileCheck --check-prefix=GFX90A %s
-// RUN: not llvm-mc -triple=amdgpu9.42 -mattr=-buffer-inv-inst -filetype=null %s 2>&1 | FileCheck --check-prefix=DISABLED %s
-
-buffer_inv sc0 sc1
-// SUPPORTED: buffer_inv sc0 sc1                    ; encoding: [0x00,0xc0,0xa4,0xe0,0x00,0x00,0x00,0x00]
-// GFX90A: error: instruction not supported on this GPU (gfx90a): buffer_inv
-// DISABLED: error: instruction not supported on this GPU (gfx942): buffer_inv
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 2847942227df33..62d6f3a4c955b4 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -2875,18 +2875,6 @@ TEST(TargetParserTest, testAMDGPUgetFeatureBitset) {
   EXPECT_TRUE(Empty.empty());
 }
 
-TEST(TargetParserTest, testAMDGPUBufferInvInstFeature) {
-  auto Has = [](AMDGPU::GPUKind AK) {
-    return AMDGPU::getFeatureBitset(AK).test(AMDGPU::FEAT_BUFFER_INV_INST);
-  };
-
-  EXPECT_FALSE(Has(AMDGPU::GK_GFX90A));
-  EXPECT_TRUE(Has(AMDGPU::GK_GFX942));
-  EXPECT_TRUE(Has(AMDGPU::GK_GFX950));
-  EXPECT_TRUE(Has(AMDGPU::GK_GFX9_4_GENERIC));
-  EXPECT_FALSE(Has(AMDGPU::GK_GFX1250));
-}
-
 TEST(TargetParserTest, testAMDGPUHalfAddressableLDSFeature) {
   auto Has = [](AMDGPU::GPUKind AK) {
     return AMDGPU::getFeatureBitset(AK).test(

>From 5baf35285b6b1e50ae9370c9512a13fb56e17ac6 Mon Sep 17 00:00:00 2001
From: Dan Zimmerman <danzimm at meta.com>
Date: Thu, 24 Sep 2026 14:48:12 -0700
Subject: [PATCH 5/5] [AMDGPU] Inherit predicates for GFX940 buffer
 invalidation instructions

---
 llvm/lib/Target/AMDGPU/BUFInstructions.td | 4 ----
 1 file changed, 4 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/BUFInstructions.td b/llvm/lib/Target/AMDGPU/BUFInstructions.td
index 2b5d20c6849f78..6cb0f77e0dbeb3 100644
--- a/llvm/lib/Target/AMDGPU/BUFInstructions.td
+++ b/llvm/lib/Target/AMDGPU/BUFInstructions.td
@@ -3660,11 +3660,7 @@ let AsmString = BUFFER_WBL2.Mnemonic, // drop flags
 defm BUFFER_WBL2  : MUBUF_Real_gfx90a<0x28>;
 defm BUFFER_INVL2 : MUBUF_Real_gfx90a<0x29>;
 
-let SubtargetPredicate = isGFX940Plus in
 def BUFFER_WBL2_gfx940  : MUBUF_Real_gfx940<0x28, BUFFER_WBL2>;
-
-let SubtargetPredicate = HasBufferInvInst,
-    AssemblerPredicate = HasBufferInvInst in
 def BUFFER_INV_gfx940   : MUBUF_Real_gfx940<0x29, BUFFER_INV>;
 
 class MTBUF_Real_Base_vi <bits<4> op, MTBUF_Pseudo ps, int Enc> :



More information about the llvm-commits mailing list