[llvm] [AMDGPU] Add the 3-dword image_gather4 variant for packed D16 + TFE (PR #215972)
Arseniy Obolenskiy via llvm-commits
llvm-commits at lists.llvm.org
Fri Aug 28 08:31:15 PDT 2026
https://github.com/aobolensk updated https://github.com/llvm/llvm-project/pull/215972
>From 2dbb32316f9cc161e761e9210c2fc2c5748e4b4d Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Thu, 13 Aug 2026 08:48:51 +0200
Subject: [PATCH 1/2] [AMDGPU] Add the 3-dword image_gather4 variant for packed
D16 + TFE
MIMG_Gather only defined V2/V4/V5 destination-dword variants, so a d16 gather4 with tfe (which needs 3 dwords on packed-D16 targets) hit "Cannot select" on gfx10+
---
llvm/lib/Target/AMDGPU/MIMGInstructions.td | 2 ++
.../AMDGPU/llvm.amdgcn.image.gather4.d16.dim.ll | 15 +++++++++++++++
2 files changed, 17 insertions(+)
diff --git a/llvm/lib/Target/AMDGPU/MIMGInstructions.td b/llvm/lib/Target/AMDGPU/MIMGInstructions.td
index cb554073f6398..3c91d340d398c 100644
--- a/llvm/lib/Target/AMDGPU/MIMGInstructions.td
+++ b/llvm/lib/Target/AMDGPU/MIMGInstructions.td
@@ -1641,6 +1641,8 @@ multiclass MIMG_Gather <mimgopc op, AMDGPUSampleVariant sample, bit wqm = 0,
Gather4 = 1 in {
let VDataDwords = 2 in
defm _V2 : MIMG_Sampler_Src_Helper<op, asm, sample, AVLdSt_64, /*enableDisasm*/ true>; /* for packed D16 only */
+ let VDataDwords = 3 in
+ defm _V3 : MIMG_Sampler_Src_Helper<op, asm, sample, AVLdSt_96>; /* packed D16 + tfe */
let VDataDwords = 4 in
defm _V4 : MIMG_Sampler_Src_Helper<op, asm, sample, AVLdSt_128>;
let VDataDwords = 5 in
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.ll
index 693344933a8e2..6f89dc9eb74bd 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.ll
@@ -23,7 +23,22 @@ main_body:
ret <2 x float> %r
}
+; GCN-LABEL: {{^}}image_gather4_b_2d_v4f16_tfe:
+; UNPACKED: image_gather4_b v[{{[0-9]+:[0-9]+}}], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 tfe d16{{$}}
+; GFX9: image_gather4_b v[0:4], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 tfe d16{{$}}
+; GFX10: image_gather4_b v[0:2], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 dim:SQ_RSRC_IMG_2D tfe d16{{$}}
+; GFX12PLUS: image_gather4_b v[0:2], [v{{[0-9]+}}, v{{[0-9]+}}, v{{[0-9]+}}], s[0:7], s[8:11] dmask:0x4 dim:SQ_RSRC_IMG_2D tfe d16{{$}}
+define amdgpu_ps <4 x half> @image_gather4_b_2d_v4f16_tfe(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %bias, float %s, float %t, ptr addrspace(1) %out) {
+main_body:
+ %r = call { <4 x half>, i32 } @llvm.amdgcn.image.gather4.b.2d.sl_v4f16i32s.f32.f32(i32 4, float %bias, float %s, float %t, <8 x i32> %rsrc, <4 x i32> %samp, i1 false, i32 1, i32 0)
+ %tex = extractvalue { <4 x half>, i32 } %r, 0
+ %tfe = extractvalue { <4 x half>, i32 } %r, 1
+ store i32 %tfe, ptr addrspace(1) %out
+ ret <4 x half> %tex
+}
+
declare <4 x half> @llvm.amdgcn.image.gather4.b.2d.v4f16.f32.f32(i32, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
+declare { <4 x half>, i32 } @llvm.amdgcn.image.gather4.b.2d.sl_v4f16i32s.f32.f32(i32, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
attributes #0 = { nounwind }
attributes #1 = { nounwind readonly }
>From f3bd71afff4626f9f476d060ff3ecaac1fdf08a6 Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Fri, 28 Aug 2026 17:31:03 +0200
Subject: [PATCH 2/2] add test
---
...lvm.amdgcn.image.gather4.d16.dim.gfx908.ll | 19 +++++++++++++++++++
1 file changed, 19 insertions(+)
create mode 100644 llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.gfx908.ll
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.gfx908.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.gfx908.ll
new file mode 100644
index 0000000000000..4d7b9b9ef8837
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.gfx908.ll
@@ -0,0 +1,19 @@
+; RUN: llc < %s -mtriple=amdgpu9.08 | FileCheck -check-prefix=GCN %s
+
+; GCN-LABEL: {{^}}image_gather4_b_2d_v4f16_tfe_agpr:
+; GCN: image_gather4_b v[{{[0-9]+:[0-9]+}}], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 tfe d16{{$}}
+; GCN: v_accvgpr_write_b32 a{{[0-9]+}}, v{{[0-9]+}}
+; GCN: v_accvgpr_write_b32 a{{[0-9]+}}, v{{[0-9]+}}
+; GCN: v_accvgpr_write_b32 a{{[0-9]+}}, v{{[0-9]+}}
+define amdgpu_ps void @image_gather4_b_2d_v4f16_tfe_agpr(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %bias, float %s, float %t) {
+main_body:
+ %r = call { <4 x half>, i32 } @llvm.amdgcn.image.gather4.b.2d.sl_v4f16i32s.f32.f32(i32 4, float %bias, float %s, float %t, <8 x i32> %rsrc, <4 x i32> %samp, i1 false, i32 1, i32 0)
+ %tex = extractvalue { <4 x half>, i32 } %r, 0
+ %tfe = extractvalue { <4 x half>, i32 } %r, 1
+ call void asm sideeffect "; use $0 $1", "a,a"(<4 x half> %tex, i32 %tfe)
+ ret void
+}
+
+declare { <4 x half>, i32 } @llvm.amdgcn.image.gather4.b.2d.sl_v4f16i32s.f32.f32(i32, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #0
+
+attributes #0 = { nounwind readonly }
More information about the llvm-commits
mailing list