[llvm] 468ef9b - [AMDGPU] Add the 3-dword image_gather4 variant for packed D16 + TFE (#215972)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Aug 28 12:20:34 PDT 2026
Author: Arseniy Obolenskiy
Date: 2026-08-28T21:20:29+02:00
New Revision: 468ef9b4c30b2524ff6f343154e150b779a48c88
URL: https://github.com/llvm/llvm-project/commit/468ef9b4c30b2524ff6f343154e150b779a48c88
DIFF: https://github.com/llvm/llvm-project/commit/468ef9b4c30b2524ff6f343154e150b779a48c88.diff
LOG: [AMDGPU] Add the 3-dword image_gather4 variant for packed D16 + TFE (#215972)
MIMG_Gather only defined V2/V4/V5 destination-dword variants, so a d16
gather4 with tfe (which needs 3 dwords on packed-D16 targets) hit
"Cannot select" on gfx10+
Added:
Modified:
llvm/lib/Target/AMDGPU/MIMGInstructions.td
llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.ll
Removed:
################################################################################
diff --git a/llvm/lib/Target/AMDGPU/MIMGInstructions.td b/llvm/lib/Target/AMDGPU/MIMGInstructions.td
index 1a4cd39cabfff..1a53d2a1898e5 100644
--- a/llvm/lib/Target/AMDGPU/MIMGInstructions.td
+++ b/llvm/lib/Target/AMDGPU/MIMGInstructions.td
@@ -1643,6 +1643,8 @@ multiclass MIMG_Gather <mimgopc op, AMDGPUSampleVariant sample, bit wqm = 0,
Gather4 = 1 in {
let VDataDwords = 2 in
defm _V2 : MIMG_Sampler_Src_Helper<op, asm, sample, AVLdSt_64, /*enableDisasm*/ true>; /* for packed D16 only */
+ let VDataDwords = 3 in
+ defm _V3 : MIMG_Sampler_Src_Helper<op, asm, sample, AVLdSt_96>; /* packed D16 + tfe */
let VDataDwords = 4 in
defm _V4 : MIMG_Sampler_Src_Helper<op, asm, sample, AVLdSt_128>;
let VDataDwords = 5 in
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.ll
index 693344933a8e2..6f89dc9eb74bd 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.image.gather4.d16.dim.ll
@@ -23,7 +23,22 @@ main_body:
ret <2 x float> %r
}
+; GCN-LABEL: {{^}}image_gather4_b_2d_v4f16_tfe:
+; UNPACKED: image_gather4_b v[{{[0-9]+:[0-9]+}}], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 tfe d16{{$}}
+; GFX9: image_gather4_b v[0:4], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 tfe d16{{$}}
+; GFX10: image_gather4_b v[0:2], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 dim:SQ_RSRC_IMG_2D tfe d16{{$}}
+; GFX12PLUS: image_gather4_b v[0:2], [v{{[0-9]+}}, v{{[0-9]+}}, v{{[0-9]+}}], s[0:7], s[8:11] dmask:0x4 dim:SQ_RSRC_IMG_2D tfe d16{{$}}
+define amdgpu_ps <4 x half> @image_gather4_b_2d_v4f16_tfe(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %bias, float %s, float %t, ptr addrspace(1) %out) {
+main_body:
+ %r = call { <4 x half>, i32 } @llvm.amdgcn.image.gather4.b.2d.sl_v4f16i32s.f32.f32(i32 4, float %bias, float %s, float %t, <8 x i32> %rsrc, <4 x i32> %samp, i1 false, i32 1, i32 0)
+ %tex = extractvalue { <4 x half>, i32 } %r, 0
+ %tfe = extractvalue { <4 x half>, i32 } %r, 1
+ store i32 %tfe, ptr addrspace(1) %out
+ ret <4 x half> %tex
+}
+
declare <4 x half> @llvm.amdgcn.image.gather4.b.2d.v4f16.f32.f32(i32, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
+declare { <4 x half>, i32 } @llvm.amdgcn.image.gather4.b.2d.sl_v4f16i32s.f32.f32(i32, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
attributes #0 = { nounwind }
attributes #1 = { nounwind readonly }
More information about the llvm-commits
mailing list