[llvm] [AMDGPU] Add the 3-dword image_gather4 variant for packed D16 + TFE (PR #215972)
Arseniy Obolenskiy via llvm-commits
llvm-commits at lists.llvm.org
Fri Aug 28 08:31:15 PDT 2026
================
@@ -23,7 +23,22 @@ main_body:
ret <2 x float> %r
}
+; GCN-LABEL: {{^}}image_gather4_b_2d_v4f16_tfe:
+; UNPACKED: image_gather4_b v[{{[0-9]+:[0-9]+}}], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 tfe d16{{$}}
+; GFX9: image_gather4_b v[0:4], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 tfe d16{{$}}
+; GFX10: image_gather4_b v[0:2], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 dim:SQ_RSRC_IMG_2D tfe d16{{$}}
+; GFX12PLUS: image_gather4_b v[0:2], [v{{[0-9]+}}, v{{[0-9]+}}, v{{[0-9]+}}], s[0:7], s[8:11] dmask:0x4 dim:SQ_RSRC_IMG_2D tfe d16{{$}}
+define amdgpu_ps <4 x half> @image_gather4_b_2d_v4f16_tfe(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %bias, float %s, float %t, ptr addrspace(1) %out) {
+main_body:
+ %r = call { <4 x half>, i32 } @llvm.amdgcn.image.gather4.b.2d.sl_v4f16i32s.f32.f32(i32 4, float %bias, float %s, float %t, <8 x i32> %rsrc, <4 x i32> %samp, i1 false, i32 1, i32 0)
+ %tex = extractvalue { <4 x half>, i32 } %r, 0
+ %tfe = extractvalue { <4 x half>, i32 } %r, 1
+ store i32 %tfe, ptr addrspace(1) %out
+ ret <4 x half> %tex
+}
+
----------------
aobolensk wrote:
Why not, done
https://github.com/llvm/llvm-project/pull/215972
More information about the llvm-commits
mailing list