[llvm] [AMDGPU] Reject SDWA forms the subtarget cannot encode (PR #218592)

Arseniy Obolenskiy via llvm-commits llvm-commits at lists.llvm.org
Tue Aug 25 05:00:52 PDT 2026


https://github.com/aobolensk updated https://github.com/llvm/llvm-project/pull/218592

>From 6acfb2edc7ff92b0fcd534b1c0878a1a5aace38a Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Tue, 25 Aug 2026 08:23:44 +0200
Subject: [PATCH 1/2] [AMDGPU] Reject SDWA forms the subtarget cannot encode

isConvertibleToSDWA only checked the base opcode, so it could still fold into an SDWA form the target cannot encode and crash later
---
 llvm/lib/Target/AMDGPU/SIPeepholeSDWA.cpp     |  3 +-
 .../CodeGen/AMDGPU/sdwa-peephole-movrels.ll   | 72 +++++++++++++++++++
 .../CodeGen/AMDGPU/sdwa-peephole-movrels.mir  | 40 +++++++++++
 3 files changed, 114 insertions(+), 1 deletion(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.ll
 create mode 100644 llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.mir

diff --git a/llvm/lib/Target/AMDGPU/SIPeepholeSDWA.cpp b/llvm/lib/Target/AMDGPU/SIPeepholeSDWA.cpp
index f8a28983d3e11..d30b192875906 100644
--- a/llvm/lib/Target/AMDGPU/SIPeepholeSDWA.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPeepholeSDWA.cpp
@@ -1179,7 +1179,8 @@ bool isConvertibleToSDWA(MachineInstr &MI,
     return false;
 
   // Check if target supports this SDWA opcode
-  if (TII->pseudoToMCOpcode(Opc) == -1)
+  if (TII->pseudoToMCOpcode(Opc) == -1 ||
+      TII->pseudoToMCOpcode(AMDGPU::getSDWAOp(Opc)) == -1)
     return false;
 
   if (MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0)) {
diff --git a/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.ll b/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.ll
new file mode 100644
index 0000000000000..f9158c6b6a211
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.ll
@@ -0,0 +1,72 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx803 < %s | FileCheck -check-prefix=GFX8 %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1010 < %s | FileCheck -check-prefix=GFX10 %s
+
+; V_MOVRELS_B32_sdwa is unencodable here, folding the shl into it used to
+; crash BranchRelaxation on the invalid pseudo.
+define amdgpu_kernel void @extract_wo_offset_shl(ptr addrspace(1) %out, i32 %in) {
+; GFX8-LABEL: extract_wo_offset_shl:
+; GFX8:       ; %bb.0:
+; GFX8-NEXT:    s_load_dwordx2 s[0:1], s[8:9], 0x0
+; GFX8-NEXT:    s_load_dword s2, s[8:9], 0x8
+; GFX8-NEXT:    v_mov_b32_e32 v0, 1
+; GFX8-NEXT:    s_add_i32 s12, s12, s17
+; GFX8-NEXT:    v_mov_b32_e32 v1, 2
+; GFX8-NEXT:    v_mov_b32_e32 v2, 3
+; GFX8-NEXT:    s_waitcnt lgkmcnt(0)
+; GFX8-NEXT:    s_mov_b32 m0, s2
+; GFX8-NEXT:    v_mov_b32_e32 v3, 4
+; GFX8-NEXT:    v_mov_b32_e32 v4, 5
+; GFX8-NEXT:    v_mov_b32_e32 v5, 6
+; GFX8-NEXT:    v_mov_b32_e32 v6, 7
+; GFX8-NEXT:    v_mov_b32_e32 v7, 8
+; GFX8-NEXT:    v_mov_b32_e32 v8, 9
+; GFX8-NEXT:    v_mov_b32_e32 v9, 10
+; GFX8-NEXT:    v_mov_b32_e32 v10, 11
+; GFX8-NEXT:    v_mov_b32_e32 v11, 12
+; GFX8-NEXT:    v_mov_b32_e32 v12, 13
+; GFX8-NEXT:    v_mov_b32_e32 v13, 14
+; GFX8-NEXT:    v_mov_b32_e32 v14, 15
+; GFX8-NEXT:    v_mov_b32_e32 v15, 16
+; GFX8-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX8-NEXT:    s_mov_b32 flat_scratch_lo, s13
+; GFX8-NEXT:    s_lshr_b32 flat_scratch_hi, s12, 8
+; GFX8-NEXT:    v_lshlrev_b32_e32 v2, 16, v0
+; GFX8-NEXT:    v_mov_b32_e32 v0, s0
+; GFX8-NEXT:    v_mov_b32_e32 v1, s1
+; GFX8-NEXT:    flat_store_dword v[0:1], v2
+; GFX8-NEXT:    s_endpgm
+;
+; GFX10-LABEL: extract_wo_offset_shl:
+; GFX10:       ; %bb.0:
+; GFX10-NEXT:    s_clause 0x1
+; GFX10-NEXT:    s_load_dword s2, s[8:9], 0x8
+; GFX10-NEXT:    s_load_dwordx2 s[0:1], s[8:9], 0x0
+; GFX10-NEXT:    v_mov_b32_e32 v0, 1
+; GFX10-NEXT:    v_mov_b32_e32 v1, 2
+; GFX10-NEXT:    v_mov_b32_e32 v2, 3
+; GFX10-NEXT:    v_mov_b32_e32 v3, 4
+; GFX10-NEXT:    v_mov_b32_e32 v4, 5
+; GFX10-NEXT:    v_mov_b32_e32 v5, 6
+; GFX10-NEXT:    v_mov_b32_e32 v6, 7
+; GFX10-NEXT:    v_mov_b32_e32 v7, 8
+; GFX10-NEXT:    v_mov_b32_e32 v8, 9
+; GFX10-NEXT:    v_mov_b32_e32 v9, 10
+; GFX10-NEXT:    v_mov_b32_e32 v10, 11
+; GFX10-NEXT:    v_mov_b32_e32 v11, 12
+; GFX10-NEXT:    v_mov_b32_e32 v12, 13
+; GFX10-NEXT:    v_mov_b32_e32 v13, 14
+; GFX10-NEXT:    v_mov_b32_e32 v14, 15
+; GFX10-NEXT:    v_mov_b32_e32 v15, 16
+; GFX10-NEXT:    s_waitcnt lgkmcnt(0)
+; GFX10-NEXT:    s_mov_b32 m0, s2
+; GFX10-NEXT:    v_movrels_b32_e32 v0, v0
+; GFX10-NEXT:    v_mov_b32_e32 v1, 0
+; GFX10-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
+; GFX10-NEXT:    global_store_dword v1, v0, s[0:1]
+; GFX10-NEXT:    s_endpgm
+  %elt = extractelement <16 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16>, i32 %in
+  %shl = shl i32 %elt, 16
+  store i32 %shl, ptr addrspace(1) %out
+  ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.mir b/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.mir
new file mode 100644
index 0000000000000..6293be0f55741
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.mir
@@ -0,0 +1,40 @@
+# RUN: llc -mtriple=amdgcn -mcpu=gfx803 -run-pass=si-peephole-sdwa -verify-machineinstrs -o - %s | FileCheck %s
+# RUN: llc -mtriple=amdgcn -mcpu=gfx900 -run-pass=si-peephole-sdwa -verify-machineinstrs -o - %s | FileCheck %s
+# RUN: llc -mtriple=amdgcn -mcpu=gfx1010 -run-pass=si-peephole-sdwa -verify-machineinstrs -o - %s | FileCheck %s
+
+# V_MOVRELS_B32_sdwa is unencodable on gfx8/gfx9, and asm-only
+# (SIInstrInfo::isAsmOnlyOpcode) on gfx10.
+
+---
+name: movrels_dst_no_sdwa
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7, $sgpr0
+
+    ; CHECK-LABEL: name: movrels_dst_no_sdwa
+    ; CHECK: V_MOVRELS_B32_e32
+    ; CHECK-NOT: V_MOVRELS_B32_sdwa
+    %0:vreg_256 = COPY $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7
+    $m0 = COPY $sgpr0
+    %1:vgpr_32 = V_MOVRELS_B32_e32 %0.sub0, implicit $m0, implicit $exec, implicit %0
+    %2:vgpr_32 = V_LSHLREV_B32_e64 16, %1, implicit $exec
+    S_ENDPGM 0, implicit %2
+...
+
+# Control: an ordinary VALU instruction still converts to SDWA.
+---
+name: or_dst_sdwa
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+
+    ; CHECK-LABEL: name: or_dst_sdwa
+    ; CHECK: V_OR_B32_sdwa
+    %0:vgpr_32 = COPY $vgpr0
+    %1:vgpr_32 = COPY $vgpr1
+    %2:vgpr_32 = V_OR_B32_e64 %0, %1, implicit $exec
+    %3:vgpr_32 = V_LSHLREV_B32_e64 16, %2, implicit $exec
+    S_ENDPGM 0, implicit %3
+...

>From 01966fd0d32da0e6d7fb11850432fce1dd889264 Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Tue, 25 Aug 2026 11:37:03 +0200
Subject: [PATCH 2/2] triples

---
 llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.ll  | 4 ++--
 llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.mir | 6 +++---
 2 files changed, 5 insertions(+), 5 deletions(-)

diff --git a/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.ll b/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.ll
index f9158c6b6a211..0888c944a0b05 100644
--- a/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.ll
+++ b/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx803 < %s | FileCheck -check-prefix=GFX8 %s
-; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1010 < %s | FileCheck -check-prefix=GFX10 %s
+; RUN: llc -mtriple=amdgpu8.03-amd-amdhsa < %s | FileCheck -check-prefix=GFX8 %s
+; RUN: llc -mtriple=amdgpu10.10-amd-amdhsa < %s | FileCheck -check-prefix=GFX10 %s
 
 ; V_MOVRELS_B32_sdwa is unencodable here, folding the shl into it used to
 ; crash BranchRelaxation on the invalid pseudo.
diff --git a/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.mir b/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.mir
index 6293be0f55741..20702fcf92b4b 100644
--- a/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.mir
+++ b/llvm/test/CodeGen/AMDGPU/sdwa-peephole-movrels.mir
@@ -1,6 +1,6 @@
-# RUN: llc -mtriple=amdgcn -mcpu=gfx803 -run-pass=si-peephole-sdwa -verify-machineinstrs -o - %s | FileCheck %s
-# RUN: llc -mtriple=amdgcn -mcpu=gfx900 -run-pass=si-peephole-sdwa -verify-machineinstrs -o - %s | FileCheck %s
-# RUN: llc -mtriple=amdgcn -mcpu=gfx1010 -run-pass=si-peephole-sdwa -verify-machineinstrs -o - %s | FileCheck %s
+# RUN: llc -mtriple=amdgpu8.03 -run-pass=si-peephole-sdwa -verify-machineinstrs -o - %s | FileCheck %s
+# RUN: llc -mtriple=amdgpu9 -run-pass=si-peephole-sdwa -verify-machineinstrs -o - %s | FileCheck %s
+# RUN: llc -mtriple=amdgpu10.10 -run-pass=si-peephole-sdwa -verify-machineinstrs -o - %s | FileCheck %s
 
 # V_MOVRELS_B32_sdwa is unencodable on gfx8/gfx9, and asm-only
 # (SIInstrInfo::isAsmOnlyOpcode) on gfx10.



More information about the llvm-commits mailing list