[llvm] [AMDGPU] Add the 3-dword image_gather4 variant for packed D16 + TFE (PR #215972)

Arseniy Obolenskiy via llvm-commits llvm-commits at lists.llvm.org
Fri Aug 28 08:31:15 PDT 2026


================
@@ -23,7 +23,22 @@ main_body:
   ret <2 x float> %r
 }
 
+; GCN-LABEL: {{^}}image_gather4_b_2d_v4f16_tfe:
+; UNPACKED: image_gather4_b v[{{[0-9]+:[0-9]+}}], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 tfe d16{{$}}
+; GFX9: image_gather4_b v[0:4], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 tfe d16{{$}}
+; GFX10: image_gather4_b v[0:2], v[{{[0-9]+:[0-9]+}}], s[0:7], s[8:11] dmask:0x4 dim:SQ_RSRC_IMG_2D tfe d16{{$}}
+; GFX12PLUS: image_gather4_b v[0:2], [v{{[0-9]+}}, v{{[0-9]+}}, v{{[0-9]+}}], s[0:7], s[8:11] dmask:0x4 dim:SQ_RSRC_IMG_2D tfe d16{{$}}
+define amdgpu_ps <4 x half> @image_gather4_b_2d_v4f16_tfe(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %bias, float %s, float %t, ptr addrspace(1) %out) {
+main_body:
+  %r = call { <4 x half>, i32 } @llvm.amdgcn.image.gather4.b.2d.sl_v4f16i32s.f32.f32(i32 4, float %bias, float %s, float %t, <8 x i32> %rsrc, <4 x i32> %samp, i1 false, i32 1, i32 0)
+  %tex = extractvalue { <4 x half>, i32 } %r, 0
+  %tfe = extractvalue { <4 x half>, i32 } %r, 1
+  store i32 %tfe, ptr addrspace(1) %out
+  ret <4 x half> %tex
+}
+
----------------
aobolensk wrote:

Why not, done

https://github.com/llvm/llvm-project/pull/215972


More information about the llvm-commits mailing list