[llvm] f3778fb - [NFC][AMDGPU] Add tests for int to fp casts of loaded odd width integers (#227904)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 02:35:49 PDT 2026
Author: Dmitry Sidorov
Date: 2026-10-02T11:35:38+02:00
New Revision: f3778fb5b18178dcb6fe02b5b5335c8159e04d83
URL: https://github.com/llvm/llvm-project/commit/f3778fb5b18178dcb6fe02b5b5335c8159e04d83
DIFF: https://github.com/llvm/llvm-project/commit/f3778fb5b18178dcb6fe02b5b5335c8159e04d83.diff
LOG: [NFC][AMDGPU] Add tests for int to fp casts of loaded odd width integers (#227904)
Assisted-by: Claude Code Opus 5
Added:
llvm/test/CodeGen/AMDGPU/itofp-odd-width-load.ll
Modified:
llvm/test/Analysis/CostModel/AMDGPU/narrow-int-to-bfloat.ll
llvm/test/Analysis/CostModel/AMDGPU/narrow-int-to-fp.ll
llvm/test/CodeGen/AMDGPU/slp-int-to-fp.ll
Removed:
################################################################################
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/narrow-int-to-bfloat.ll b/llvm/test/Analysis/CostModel/AMDGPU/narrow-int-to-bfloat.ll
index 0e15b10325bc3..d49ab6541ed09 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/narrow-int-to-bfloat.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/narrow-int-to-bfloat.ll
@@ -239,3 +239,133 @@ define void @narrow_scalars_to_bfloat(i8 %a, i12 %b, i16 %c, i17 %d, i31 %e) {
%eu = uitofp i31 %e to bfloat
ret void
}
+
+define void @loaded_scalars_to_bfloat(ptr addrspace(1) %p) {
+; GFX6-LABEL: 'loaded_scalars_to_bfloat'
+; GFX6: Cost Model: Found an estimated cost of 1 for instruction: %as = sitofp i24 %a to bfloat
+; GFX6: Cost Model: Found an estimated cost of 1 for instruction: %bu = uitofp i24 %b to bfloat
+; GFX6: Cost Model: Found an estimated cost of 2 for instruction: %cs = sitofp i16 %c to bfloat
+;
+; GFX8-LABEL: 'loaded_scalars_to_bfloat'
+; GFX8: Cost Model: Found an estimated cost of 1 for instruction: %as = sitofp i24 %a to bfloat
+; GFX8: Cost Model: Found an estimated cost of 1 for instruction: %bu = uitofp i24 %b to bfloat
+; GFX8: Cost Model: Found an estimated cost of 8 for instruction: %cs = sitofp i16 %c to bfloat
+;
+; GFX9-LABEL: 'loaded_scalars_to_bfloat'
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %as = sitofp i24 %a to bfloat
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %bu = uitofp i24 %b to bfloat
+; GFX9: Cost Model: Found an estimated cost of 7 for instruction: %cs = sitofp i16 %c to bfloat
+;
+; NOSDWA-LABEL: 'loaded_scalars_to_bfloat'
+; NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %as = sitofp i24 %a to bfloat
+; NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %bu = uitofp i24 %b to bfloat
+; NOSDWA: Cost Model: Found an estimated cost of 7 for instruction: %cs = sitofp i16 %c to bfloat
+;
+; GFX950-LABEL: 'loaded_scalars_to_bfloat'
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %as = sitofp i24 %a to bfloat
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %bu = uitofp i24 %b to bfloat
+; GFX950: Cost Model: Found an estimated cost of 2 for instruction: %cs = sitofp i16 %c to bfloat
+;
+; GFX1250-LABEL: 'loaded_scalars_to_bfloat'
+; GFX1250: Cost Model: Found an estimated cost of 1 for instruction: %as = sitofp i24 %a to bfloat
+; GFX1250: Cost Model: Found an estimated cost of 1 for instruction: %bu = uitofp i24 %b to bfloat
+; GFX1250: Cost Model: Found an estimated cost of 2 for instruction: %cs = sitofp i16 %c to bfloat
+;
+ %a = load i24, ptr addrspace(1) %p
+ %as = sitofp i24 %a to bfloat
+ %b = load i24, ptr addrspace(1) %p
+ %bu = uitofp i24 %b to bfloat
+ %c = load i16, ptr addrspace(1) %p
+ %cs = sitofp i16 %c to bfloat
+ ret void
+}
+
+define void @loaded_wide_scalars_to_bfloat(ptr addrspace(1) %p) {
+; GFX6-LABEL: 'loaded_wide_scalars_to_bfloat'
+; GFX6: Cost Model: Found an estimated cost of 15 for instruction: %as = sitofp i40 %a to bfloat
+; GFX6: Cost Model: Found an estimated cost of 10 for instruction: %bu = uitofp i56 %b to bfloat
+; GFX6: Cost Model: Found an estimated cost of 15 for instruction: %cs = sitofp i48 %c to bfloat
+;
+; GFX8-LABEL: 'loaded_wide_scalars_to_bfloat'
+; GFX8: Cost Model: Found an estimated cost of 21 for instruction: %as = sitofp i40 %a to bfloat
+; GFX8: Cost Model: Found an estimated cost of 16 for instruction: %bu = uitofp i56 %b to bfloat
+; GFX8: Cost Model: Found an estimated cost of 21 for instruction: %cs = sitofp i48 %c to bfloat
+;
+; GFX9-LABEL: 'loaded_wide_scalars_to_bfloat'
+; GFX9: Cost Model: Found an estimated cost of 20 for instruction: %as = sitofp i40 %a to bfloat
+; GFX9: Cost Model: Found an estimated cost of 15 for instruction: %bu = uitofp i56 %b to bfloat
+; GFX9: Cost Model: Found an estimated cost of 20 for instruction: %cs = sitofp i48 %c to bfloat
+;
+; NOSDWA-LABEL: 'loaded_wide_scalars_to_bfloat'
+; NOSDWA: Cost Model: Found an estimated cost of 20 for instruction: %as = sitofp i40 %a to bfloat
+; NOSDWA: Cost Model: Found an estimated cost of 15 for instruction: %bu = uitofp i56 %b to bfloat
+; NOSDWA: Cost Model: Found an estimated cost of 20 for instruction: %cs = sitofp i48 %c to bfloat
+;
+; GFX950-LABEL: 'loaded_wide_scalars_to_bfloat'
+; GFX950: Cost Model: Found an estimated cost of 15 for instruction: %as = sitofp i40 %a to bfloat
+; GFX950: Cost Model: Found an estimated cost of 10 for instruction: %bu = uitofp i56 %b to bfloat
+; GFX950: Cost Model: Found an estimated cost of 15 for instruction: %cs = sitofp i48 %c to bfloat
+;
+; GFX1250-LABEL: 'loaded_wide_scalars_to_bfloat'
+; GFX1250: Cost Model: Found an estimated cost of 15 for instruction: %as = sitofp i40 %a to bfloat
+; GFX1250: Cost Model: Found an estimated cost of 10 for instruction: %bu = uitofp i56 %b to bfloat
+; GFX1250: Cost Model: Found an estimated cost of 15 for instruction: %cs = sitofp i48 %c to bfloat
+;
+ %a = load i40, ptr addrspace(1) %p
+ %as = sitofp i40 %a to bfloat
+ %b = load i56, ptr addrspace(1) %p
+ %bu = uitofp i56 %b to bfloat
+ %c = load volatile i48, ptr addrspace(1) %p
+ %cs = sitofp i48 %c to bfloat
+ ret void
+}
+
+define void @loaded_constant_scalars_to_bfloat(ptr addrspace(4) %p, ptr addrspace(1) %r) {
+; GFX6-LABEL: 'loaded_constant_scalars_to_bfloat'
+; GFX6: Cost Model: Found an estimated cost of 1 for instruction: %as = sitofp i24 %a to bfloat
+; GFX6: Cost Model: Found an estimated cost of 1 for instruction: %bs = sitofp i24 %b to bfloat
+; GFX6: Cost Model: Found an estimated cost of 10 for instruction: %cu = uitofp i48 %c to bfloat
+; GFX6: Cost Model: Found an estimated cost of 10 for instruction: %du = uitofp i48 %d to bfloat
+;
+; GFX8-LABEL: 'loaded_constant_scalars_to_bfloat'
+; GFX8: Cost Model: Found an estimated cost of 1 for instruction: %as = sitofp i24 %a to bfloat
+; GFX8: Cost Model: Found an estimated cost of 1 for instruction: %bs = sitofp i24 %b to bfloat
+; GFX8: Cost Model: Found an estimated cost of 16 for instruction: %cu = uitofp i48 %c to bfloat
+; GFX8: Cost Model: Found an estimated cost of 16 for instruction: %du = uitofp i48 %d to bfloat
+;
+; GFX9-LABEL: 'loaded_constant_scalars_to_bfloat'
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %as = sitofp i24 %a to bfloat
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %bs = sitofp i24 %b to bfloat
+; GFX9: Cost Model: Found an estimated cost of 15 for instruction: %cu = uitofp i48 %c to bfloat
+; GFX9: Cost Model: Found an estimated cost of 15 for instruction: %du = uitofp i48 %d to bfloat
+;
+; NOSDWA-LABEL: 'loaded_constant_scalars_to_bfloat'
+; NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %as = sitofp i24 %a to bfloat
+; NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %bs = sitofp i24 %b to bfloat
+; NOSDWA: Cost Model: Found an estimated cost of 15 for instruction: %cu = uitofp i48 %c to bfloat
+; NOSDWA: Cost Model: Found an estimated cost of 15 for instruction: %du = uitofp i48 %d to bfloat
+;
+; GFX950-LABEL: 'loaded_constant_scalars_to_bfloat'
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %as = sitofp i24 %a to bfloat
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %bs = sitofp i24 %b to bfloat
+; GFX950: Cost Model: Found an estimated cost of 10 for instruction: %cu = uitofp i48 %c to bfloat
+; GFX950: Cost Model: Found an estimated cost of 10 for instruction: %du = uitofp i48 %d to bfloat
+;
+; GFX1250-LABEL: 'loaded_constant_scalars_to_bfloat'
+; GFX1250: Cost Model: Found an estimated cost of 1 for instruction: %as = sitofp i24 %a to bfloat
+; GFX1250: Cost Model: Found an estimated cost of 1 for instruction: %bs = sitofp i24 %b to bfloat
+; GFX1250: Cost Model: Found an estimated cost of 10 for instruction: %cu = uitofp i48 %c to bfloat
+; GFX1250: Cost Model: Found an estimated cost of 10 for instruction: %du = uitofp i48 %d to bfloat
+;
+ %a = load i24, ptr addrspace(4) %p, align 4
+ %as = sitofp i24 %a to bfloat
+ %b = load i24, ptr addrspace(4) %p, align 2
+ %bs = sitofp i24 %b to bfloat
+ %c = load i48, ptr addrspace(1) %r, align 8, !invariant.load !0
+ %cu = uitofp i48 %c to bfloat
+ %d = load i48, ptr addrspace(4) %p, align 2
+ %du = uitofp i48 %d to bfloat
+ ret void
+}
+
+!0 = !{}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/narrow-int-to-fp.ll b/llvm/test/Analysis/CostModel/AMDGPU/narrow-int-to-fp.ll
index 8f4ec6ddbffd0..36cfa0e5fb5c6 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/narrow-int-to-fp.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/narrow-int-to-fp.ll
@@ -702,3 +702,435 @@ define void @wide_scalars_to_fp(i16 %a, i17 %b, i31 %c, i12 %d, i32 %e, i8 %f) {
%fud = uitofp i8 %f to double
ret void
}
+
+define void @loaded_scalars_to_fp(ptr addrspace(1) %p) {
+; NO16BIT-LABEL: 'loaded_scalars_to_fp'
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %buh = uitofp i24 %b to half
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %csf = sitofp i24 %c to float
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %duf = uitofp i24 %d to float
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %esd = sitofp i24 %e to double
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %fud = uitofp i24 %f to double
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %gsh = sitofp i24 %g to half
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %guh = uitofp i24 %g to half
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %hsh = sitofp i24 %h to half
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %ish = sitofp i16 %i to half
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %jsh = sitofp i8 %j to half
+;
+; GFX9-LABEL: 'loaded_scalars_to_fp'
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %buh = uitofp i24 %b to half
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %csf = sitofp i24 %c to float
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %duf = uitofp i24 %d to float
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %esd = sitofp i24 %e to double
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %fud = uitofp i24 %f to double
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %gsh = sitofp i24 %g to half
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %guh = uitofp i24 %g to half
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %hsh = sitofp i24 %h to half
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %ish = sitofp i16 %i to half
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %jsh = sitofp i8 %j to half
+;
+; GFX9-NOSDWA-LABEL: 'loaded_scalars_to_fp'
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %buh = uitofp i24 %b to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %csf = sitofp i24 %c to float
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %duf = uitofp i24 %d to float
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %esd = sitofp i24 %e to double
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %fud = uitofp i24 %f to double
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %gsh = sitofp i24 %g to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %guh = uitofp i24 %g to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %hsh = sitofp i24 %h to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %ish = sitofp i16 %i to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %jsh = sitofp i8 %j to half
+;
+; GFX11-FAKE16-LABEL: 'loaded_scalars_to_fp'
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %buh = uitofp i24 %b to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %csf = sitofp i24 %c to float
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %duf = uitofp i24 %d to float
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %esd = sitofp i24 %e to double
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %fud = uitofp i24 %f to double
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %gsh = sitofp i24 %g to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %guh = uitofp i24 %g to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %hsh = sitofp i24 %h to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %ish = sitofp i16 %i to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %jsh = sitofp i8 %j to half
+;
+; GFX11-TRUE16-LABEL: 'loaded_scalars_to_fp'
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %buh = uitofp i24 %b to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %csf = sitofp i24 %c to float
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %duf = uitofp i24 %d to float
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %esd = sitofp i24 %e to double
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %fud = uitofp i24 %f to double
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %gsh = sitofp i24 %g to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %guh = uitofp i24 %g to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %hsh = sitofp i24 %h to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %ish = sitofp i16 %i to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %jsh = sitofp i8 %j to half
+;
+; GFX950-LABEL: 'loaded_scalars_to_fp'
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %buh = uitofp i24 %b to half
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %csf = sitofp i24 %c to float
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %duf = uitofp i24 %d to float
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %esd = sitofp i24 %e to double
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %fud = uitofp i24 %f to double
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %gsh = sitofp i24 %g to half
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %guh = uitofp i24 %g to half
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %hsh = sitofp i24 %h to half
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %ish = sitofp i16 %i to half
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %jsh = sitofp i8 %j to half
+;
+; GFX1250-FAKE16-LABEL: 'loaded_scalars_to_fp'
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %buh = uitofp i24 %b to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %csf = sitofp i24 %c to float
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %duf = uitofp i24 %d to float
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %esd = sitofp i24 %e to double
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %fud = uitofp i24 %f to double
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %gsh = sitofp i24 %g to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %guh = uitofp i24 %g to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %hsh = sitofp i24 %h to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %ish = sitofp i16 %i to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %jsh = sitofp i8 %j to half
+;
+; GFX1250-TRUE16-LABEL: 'loaded_scalars_to_fp'
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %buh = uitofp i24 %b to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %csf = sitofp i24 %c to float
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %duf = uitofp i24 %d to float
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %esd = sitofp i24 %e to double
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %fud = uitofp i24 %f to double
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %gsh = sitofp i24 %g to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %guh = uitofp i24 %g to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %hsh = sitofp i24 %h to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %ish = sitofp i16 %i to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %jsh = sitofp i8 %j to half
+;
+; GFX6-SIZE-LABEL: 'loaded_scalars_to_fp'
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %buh = uitofp i24 %b to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %csf = sitofp i24 %c to float
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %duf = uitofp i24 %d to float
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %esd = sitofp i24 %e to double
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %fud = uitofp i24 %f to double
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %gsh = sitofp i24 %g to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %guh = uitofp i24 %g to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %hsh = sitofp i24 %h to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %ish = sitofp i16 %i to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %jsh = sitofp i8 %j to half
+;
+; GFX9-SIZE-LABEL: 'loaded_scalars_to_fp'
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %buh = uitofp i24 %b to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %csf = sitofp i24 %c to float
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %duf = uitofp i24 %d to float
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %esd = sitofp i24 %e to double
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %fud = uitofp i24 %f to double
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %gsh = sitofp i24 %g to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %guh = uitofp i24 %g to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %hsh = sitofp i24 %h to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %ish = sitofp i16 %i to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %jsh = sitofp i8 %j to half
+;
+ %a = load i24, ptr addrspace(1) %p
+ %ash = sitofp i24 %a to half
+ %b = load i24, ptr addrspace(1) %p
+ %buh = uitofp i24 %b to half
+ %c = load i24, ptr addrspace(1) %p
+ %csf = sitofp i24 %c to float
+ %d = load i24, ptr addrspace(1) %p
+ %duf = uitofp i24 %d to float
+ %e = load i24, ptr addrspace(1) %p
+ %esd = sitofp i24 %e to double
+ %f = load i24, ptr addrspace(1) %p
+ %fud = uitofp i24 %f to double
+ %g = load i24, ptr addrspace(1) %p
+ %gsh = sitofp i24 %g to half
+ %guh = uitofp i24 %g to half
+ %h = load volatile i24, ptr addrspace(1) %p
+ %hsh = sitofp i24 %h to half
+ %i = load i16, ptr addrspace(1) %p
+ %ish = sitofp i16 %i to half
+ %j = load i8, ptr addrspace(1) %p
+ %jsh = sitofp i8 %j to half
+ ret void
+}
+
+define void @loaded_wide_scalars_to_fp(ptr addrspace(1) %p) {
+; NO16BIT-LABEL: 'loaded_wide_scalars_to_fp'
+; NO16BIT: Cost Model: Found an estimated cost of 15 for instruction: %ash = sitofp i40 %a to half
+; NO16BIT: Cost Model: Found an estimated cost of 10 for instruction: %buh = uitofp i48 %b to half
+; NO16BIT: Cost Model: Found an estimated cost of 14 for instruction: %csf = sitofp i56 %c to float
+; NO16BIT: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i40 %d to float
+; NO16BIT: Cost Model: Found an estimated cost of 18 for instruction: %esd = sitofp i48 %e to double
+; NO16BIT: Cost Model: Found an estimated cost of 17 for instruction: %fud = uitofp i56 %f to double
+; NO16BIT: Cost Model: Found an estimated cost of 15 for instruction: %gsh = sitofp i48 %g to half
+; NO16BIT: Cost Model: Found an estimated cost of 10 for instruction: %guh = uitofp i48 %g to half
+; NO16BIT: Cost Model: Found an estimated cost of 15 for instruction: %hsh = sitofp i48 %h to half
+; NO16BIT: Cost Model: Found an estimated cost of 15 for instruction: %ish = sitofp i36 %i to half
+; NO16BIT: Cost Model: Found an estimated cost of 30 for instruction: %jsh = sitofp <2 x i48> %j to <2 x half>
+;
+; GFX9-LABEL: 'loaded_wide_scalars_to_fp'
+; GFX9: Cost Model: Found an estimated cost of 15 for instruction: %ash = sitofp i40 %a to half
+; GFX9: Cost Model: Found an estimated cost of 10 for instruction: %buh = uitofp i48 %b to half
+; GFX9: Cost Model: Found an estimated cost of 14 for instruction: %csf = sitofp i56 %c to float
+; GFX9: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i40 %d to float
+; GFX9: Cost Model: Found an estimated cost of 18 for instruction: %esd = sitofp i48 %e to double
+; GFX9: Cost Model: Found an estimated cost of 17 for instruction: %fud = uitofp i56 %f to double
+; GFX9: Cost Model: Found an estimated cost of 15 for instruction: %gsh = sitofp i48 %g to half
+; GFX9: Cost Model: Found an estimated cost of 10 for instruction: %guh = uitofp i48 %g to half
+; GFX9: Cost Model: Found an estimated cost of 15 for instruction: %hsh = sitofp i48 %h to half
+; GFX9: Cost Model: Found an estimated cost of 15 for instruction: %ish = sitofp i36 %i to half
+; GFX9: Cost Model: Found an estimated cost of 30 for instruction: %jsh = sitofp <2 x i48> %j to <2 x half>
+;
+; GFX9-NOSDWA-LABEL: 'loaded_wide_scalars_to_fp'
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 15 for instruction: %ash = sitofp i40 %a to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 10 for instruction: %buh = uitofp i48 %b to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 14 for instruction: %csf = sitofp i56 %c to float
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i40 %d to float
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 18 for instruction: %esd = sitofp i48 %e to double
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 17 for instruction: %fud = uitofp i56 %f to double
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 15 for instruction: %gsh = sitofp i48 %g to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 10 for instruction: %guh = uitofp i48 %g to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 15 for instruction: %hsh = sitofp i48 %h to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 15 for instruction: %ish = sitofp i36 %i to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 30 for instruction: %jsh = sitofp <2 x i48> %j to <2 x half>
+;
+; GFX11-FAKE16-LABEL: 'loaded_wide_scalars_to_fp'
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 15 for instruction: %ash = sitofp i40 %a to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 10 for instruction: %buh = uitofp i48 %b to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 14 for instruction: %csf = sitofp i56 %c to float
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i40 %d to float
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 18 for instruction: %esd = sitofp i48 %e to double
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 17 for instruction: %fud = uitofp i56 %f to double
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 15 for instruction: %gsh = sitofp i48 %g to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 10 for instruction: %guh = uitofp i48 %g to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 15 for instruction: %hsh = sitofp i48 %h to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 15 for instruction: %ish = sitofp i36 %i to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 30 for instruction: %jsh = sitofp <2 x i48> %j to <2 x half>
+;
+; GFX11-TRUE16-LABEL: 'loaded_wide_scalars_to_fp'
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 15 for instruction: %ash = sitofp i40 %a to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 10 for instruction: %buh = uitofp i48 %b to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 14 for instruction: %csf = sitofp i56 %c to float
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i40 %d to float
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 18 for instruction: %esd = sitofp i48 %e to double
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 17 for instruction: %fud = uitofp i56 %f to double
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 15 for instruction: %gsh = sitofp i48 %g to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 10 for instruction: %guh = uitofp i48 %g to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 15 for instruction: %hsh = sitofp i48 %h to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 15 for instruction: %ish = sitofp i36 %i to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 30 for instruction: %jsh = sitofp <2 x i48> %j to <2 x half>
+;
+; GFX950-LABEL: 'loaded_wide_scalars_to_fp'
+; GFX950: Cost Model: Found an estimated cost of 15 for instruction: %ash = sitofp i40 %a to half
+; GFX950: Cost Model: Found an estimated cost of 10 for instruction: %buh = uitofp i48 %b to half
+; GFX950: Cost Model: Found an estimated cost of 14 for instruction: %csf = sitofp i56 %c to float
+; GFX950: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i40 %d to float
+; GFX950: Cost Model: Found an estimated cost of 6 for instruction: %esd = sitofp i48 %e to double
+; GFX950: Cost Model: Found an estimated cost of 5 for instruction: %fud = uitofp i56 %f to double
+; GFX950: Cost Model: Found an estimated cost of 15 for instruction: %gsh = sitofp i48 %g to half
+; GFX950: Cost Model: Found an estimated cost of 10 for instruction: %guh = uitofp i48 %g to half
+; GFX950: Cost Model: Found an estimated cost of 15 for instruction: %hsh = sitofp i48 %h to half
+; GFX950: Cost Model: Found an estimated cost of 15 for instruction: %ish = sitofp i36 %i to half
+; GFX950: Cost Model: Found an estimated cost of 30 for instruction: %jsh = sitofp <2 x i48> %j to <2 x half>
+;
+; GFX1250-FAKE16-LABEL: 'loaded_wide_scalars_to_fp'
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 15 for instruction: %ash = sitofp i40 %a to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 10 for instruction: %buh = uitofp i48 %b to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 14 for instruction: %csf = sitofp i56 %c to float
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i40 %d to float
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 18 for instruction: %esd = sitofp i48 %e to double
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 17 for instruction: %fud = uitofp i56 %f to double
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 15 for instruction: %gsh = sitofp i48 %g to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 10 for instruction: %guh = uitofp i48 %g to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 15 for instruction: %hsh = sitofp i48 %h to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 15 for instruction: %ish = sitofp i36 %i to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 30 for instruction: %jsh = sitofp <2 x i48> %j to <2 x half>
+;
+; GFX1250-TRUE16-LABEL: 'loaded_wide_scalars_to_fp'
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 15 for instruction: %ash = sitofp i40 %a to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 10 for instruction: %buh = uitofp i48 %b to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 14 for instruction: %csf = sitofp i56 %c to float
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i40 %d to float
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 18 for instruction: %esd = sitofp i48 %e to double
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 17 for instruction: %fud = uitofp i56 %f to double
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 15 for instruction: %gsh = sitofp i48 %g to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 10 for instruction: %guh = uitofp i48 %g to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 15 for instruction: %hsh = sitofp i48 %h to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 15 for instruction: %ish = sitofp i36 %i to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 30 for instruction: %jsh = sitofp <2 x i48> %j to <2 x half>
+;
+; GFX6-SIZE-LABEL: 'loaded_wide_scalars_to_fp'
+; GFX6-SIZE: Cost Model: Found an estimated cost of 15 for instruction: %ash = sitofp i40 %a to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 10 for instruction: %buh = uitofp i48 %b to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 14 for instruction: %csf = sitofp i56 %c to float
+; GFX6-SIZE: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i40 %d to float
+; GFX6-SIZE: Cost Model: Found an estimated cost of 10 for instruction: %esd = sitofp i48 %e to double
+; GFX6-SIZE: Cost Model: Found an estimated cost of 9 for instruction: %fud = uitofp i56 %f to double
+; GFX6-SIZE: Cost Model: Found an estimated cost of 15 for instruction: %gsh = sitofp i48 %g to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 10 for instruction: %guh = uitofp i48 %g to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 15 for instruction: %hsh = sitofp i48 %h to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 15 for instruction: %ish = sitofp i36 %i to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 30 for instruction: %jsh = sitofp <2 x i48> %j to <2 x half>
+;
+; GFX9-SIZE-LABEL: 'loaded_wide_scalars_to_fp'
+; GFX9-SIZE: Cost Model: Found an estimated cost of 15 for instruction: %ash = sitofp i40 %a to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 10 for instruction: %buh = uitofp i48 %b to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 14 for instruction: %csf = sitofp i56 %c to float
+; GFX9-SIZE: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i40 %d to float
+; GFX9-SIZE: Cost Model: Found an estimated cost of 10 for instruction: %esd = sitofp i48 %e to double
+; GFX9-SIZE: Cost Model: Found an estimated cost of 9 for instruction: %fud = uitofp i56 %f to double
+; GFX9-SIZE: Cost Model: Found an estimated cost of 15 for instruction: %gsh = sitofp i48 %g to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 10 for instruction: %guh = uitofp i48 %g to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 15 for instruction: %hsh = sitofp i48 %h to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 15 for instruction: %ish = sitofp i36 %i to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 30 for instruction: %jsh = sitofp <2 x i48> %j to <2 x half>
+;
+ %a = load i40, ptr addrspace(1) %p
+ %ash = sitofp i40 %a to half
+ %b = load i48, ptr addrspace(1) %p
+ %buh = uitofp i48 %b to half
+ %c = load i56, ptr addrspace(1) %p
+ %csf = sitofp i56 %c to float
+ %d = load i40, ptr addrspace(1) %p
+ %duf = uitofp i40 %d to float
+ %e = load i48, ptr addrspace(1) %p
+ %esd = sitofp i48 %e to double
+ %f = load i56, ptr addrspace(1) %p
+ %fud = uitofp i56 %f to double
+ %g = load i48, ptr addrspace(1) %p
+ %gsh = sitofp i48 %g to half
+ %guh = uitofp i48 %g to half
+ %h = load volatile i48, ptr addrspace(1) %p
+ %hsh = sitofp i48 %h to half
+ %i = load i36, ptr addrspace(1) %p
+ %ish = sitofp i36 %i to half
+ %j = load <2 x i48>, ptr addrspace(1) %p
+ %jsh = sitofp <2 x i48> %j to <2 x half>
+ ret void
+}
+
+define void @loaded_constant_scalars_to_fp(ptr addrspace(4) %p, ptr addrspace(6) %q, ptr addrspace(1) %r) {
+; NO16BIT-LABEL: 'loaded_constant_scalars_to_fp'
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %bsh = sitofp i24 %b to half
+; NO16BIT: Cost Model: Found an estimated cost of 9 for instruction: %cuf = uitofp i48 %c to float
+; NO16BIT: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i48 %d to float
+; NO16BIT: Cost Model: Found an estimated cost of 15 for instruction: %esh = sitofp i40 %e to half
+; NO16BIT: Cost Model: Found an estimated cost of 14 for instruction: %fsf = sitofp i56 %f to float
+; NO16BIT: Cost Model: Found an estimated cost of 14 for instruction: %gsf = sitofp i56 %g to float
+; NO16BIT: Cost Model: Found an estimated cost of 1 for instruction: %hsd = sitofp i24 %h to double
+;
+; GFX9-LABEL: 'loaded_constant_scalars_to_fp'
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %bsh = sitofp i24 %b to half
+; GFX9: Cost Model: Found an estimated cost of 9 for instruction: %cuf = uitofp i48 %c to float
+; GFX9: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i48 %d to float
+; GFX9: Cost Model: Found an estimated cost of 15 for instruction: %esh = sitofp i40 %e to half
+; GFX9: Cost Model: Found an estimated cost of 14 for instruction: %fsf = sitofp i56 %f to float
+; GFX9: Cost Model: Found an estimated cost of 14 for instruction: %gsf = sitofp i56 %g to float
+; GFX9: Cost Model: Found an estimated cost of 1 for instruction: %hsd = sitofp i24 %h to double
+;
+; GFX9-NOSDWA-LABEL: 'loaded_constant_scalars_to_fp'
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %bsh = sitofp i24 %b to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 9 for instruction: %cuf = uitofp i48 %c to float
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i48 %d to float
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 15 for instruction: %esh = sitofp i40 %e to half
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 14 for instruction: %fsf = sitofp i56 %f to float
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 14 for instruction: %gsf = sitofp i56 %g to float
+; GFX9-NOSDWA: Cost Model: Found an estimated cost of 1 for instruction: %hsd = sitofp i24 %h to double
+;
+; GFX11-FAKE16-LABEL: 'loaded_constant_scalars_to_fp'
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %bsh = sitofp i24 %b to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 9 for instruction: %cuf = uitofp i48 %c to float
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i48 %d to float
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 15 for instruction: %esh = sitofp i40 %e to half
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 14 for instruction: %fsf = sitofp i56 %f to float
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 14 for instruction: %gsf = sitofp i56 %g to float
+; GFX11-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %hsd = sitofp i24 %h to double
+;
+; GFX11-TRUE16-LABEL: 'loaded_constant_scalars_to_fp'
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %bsh = sitofp i24 %b to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 9 for instruction: %cuf = uitofp i48 %c to float
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i48 %d to float
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 15 for instruction: %esh = sitofp i40 %e to half
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 14 for instruction: %fsf = sitofp i56 %f to float
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 14 for instruction: %gsf = sitofp i56 %g to float
+; GFX11-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %hsd = sitofp i24 %h to double
+;
+; GFX950-LABEL: 'loaded_constant_scalars_to_fp'
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %bsh = sitofp i24 %b to half
+; GFX950: Cost Model: Found an estimated cost of 9 for instruction: %cuf = uitofp i48 %c to float
+; GFX950: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i48 %d to float
+; GFX950: Cost Model: Found an estimated cost of 15 for instruction: %esh = sitofp i40 %e to half
+; GFX950: Cost Model: Found an estimated cost of 14 for instruction: %fsf = sitofp i56 %f to float
+; GFX950: Cost Model: Found an estimated cost of 14 for instruction: %gsf = sitofp i56 %g to float
+; GFX950: Cost Model: Found an estimated cost of 1 for instruction: %hsd = sitofp i24 %h to double
+;
+; GFX1250-FAKE16-LABEL: 'loaded_constant_scalars_to_fp'
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %bsh = sitofp i24 %b to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 9 for instruction: %cuf = uitofp i48 %c to float
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i48 %d to float
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 15 for instruction: %esh = sitofp i40 %e to half
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 14 for instruction: %fsf = sitofp i56 %f to float
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 14 for instruction: %gsf = sitofp i56 %g to float
+; GFX1250-FAKE16: Cost Model: Found an estimated cost of 1 for instruction: %hsd = sitofp i24 %h to double
+;
+; GFX1250-TRUE16-LABEL: 'loaded_constant_scalars_to_fp'
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %bsh = sitofp i24 %b to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 9 for instruction: %cuf = uitofp i48 %c to float
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i48 %d to float
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 15 for instruction: %esh = sitofp i40 %e to half
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 14 for instruction: %fsf = sitofp i56 %f to float
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 14 for instruction: %gsf = sitofp i56 %g to float
+; GFX1250-TRUE16: Cost Model: Found an estimated cost of 1 for instruction: %hsd = sitofp i24 %h to double
+;
+; GFX6-SIZE-LABEL: 'loaded_constant_scalars_to_fp'
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %bsh = sitofp i24 %b to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 9 for instruction: %cuf = uitofp i48 %c to float
+; GFX6-SIZE: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i48 %d to float
+; GFX6-SIZE: Cost Model: Found an estimated cost of 15 for instruction: %esh = sitofp i40 %e to half
+; GFX6-SIZE: Cost Model: Found an estimated cost of 14 for instruction: %fsf = sitofp i56 %f to float
+; GFX6-SIZE: Cost Model: Found an estimated cost of 14 for instruction: %gsf = sitofp i56 %g to float
+; GFX6-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %hsd = sitofp i24 %h to double
+;
+; GFX9-SIZE-LABEL: 'loaded_constant_scalars_to_fp'
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %ash = sitofp i24 %a to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %bsh = sitofp i24 %b to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 9 for instruction: %cuf = uitofp i48 %c to float
+; GFX9-SIZE: Cost Model: Found an estimated cost of 9 for instruction: %duf = uitofp i48 %d to float
+; GFX9-SIZE: Cost Model: Found an estimated cost of 15 for instruction: %esh = sitofp i40 %e to half
+; GFX9-SIZE: Cost Model: Found an estimated cost of 14 for instruction: %fsf = sitofp i56 %f to float
+; GFX9-SIZE: Cost Model: Found an estimated cost of 14 for instruction: %gsf = sitofp i56 %g to float
+; GFX9-SIZE: Cost Model: Found an estimated cost of 1 for instruction: %hsd = sitofp i24 %h to double
+;
+ %a = load i24, ptr addrspace(4) %p, align 4
+ %ash = sitofp i24 %a to half
+ %b = load i24, ptr addrspace(4) %p, align 2
+ %bsh = sitofp i24 %b to half
+ %c = load i48, ptr addrspace(4) %p, align 8
+ %cuf = uitofp i48 %c to float
+ %d = load i48, ptr addrspace(4) %p, align 2
+ %duf = uitofp i48 %d to float
+ %e = load i40, ptr addrspace(6) %q, align 4
+ %esh = sitofp i40 %e to half
+ %f = load i56, ptr addrspace(1) %r, align 8, !invariant.load !0
+ %fsf = sitofp i56 %f to float
+ %g = load i56, ptr addrspace(1) %r, align 2, !invariant.load !0
+ %gsf = sitofp i56 %g to float
+ %h = load i24, ptr addrspace(1) %r, align 4, !invariant.load !0
+ %hsd = sitofp i24 %h to double
+ ret void
+}
+
+!0 = !{}
diff --git a/llvm/test/CodeGen/AMDGPU/itofp-odd-width-load.ll b/llvm/test/CodeGen/AMDGPU/itofp-odd-width-load.ll
new file mode 100644
index 0000000000000..75dd8d6748264
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/itofp-odd-width-load.ll
@@ -0,0 +1,466 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefix=GFX900 %s
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa < %s | FileCheck -check-prefix=GFX1250 %s
+
+; A uniform constant or invariant load of 24 or 48 bits aligned to 4 bytes is
+; widened and keeps its extension. Other loads are split and extend in the
+; high part load.
+
+define amdgpu_kernel void @sitofp_i24_constant_align4(ptr addrspace(4) %in, ptr addrspace(1) %out) {
+; GFX900-LABEL: sitofp_i24_constant_align4:
+; GFX900: ; %bb.0:
+; GFX900-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX900-NEXT: v_mov_b32_e32 v1, 0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: s_load_dword s0, s[0:1], 0x0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: s_bfe_i32 s0, s0, 0x180000
+; GFX900-NEXT: v_cvt_f32_i32_e32 v0, s0
+; GFX900-NEXT: global_store_dword v1, v0, s[2:3]
+; GFX900-NEXT: s_endpgm
+;
+; GFX1250-LABEL: sitofp_i24_constant_align4:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT: s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_load_b32 s0, s[0:1], 0x0 nv
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_bfe_i32 s0, s0, 0x180000
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
+; GFX1250-NEXT: s_cvt_f32_i32 s0, s0
+; GFX1250-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s0
+; GFX1250-NEXT: global_store_b32 v0, v1, s[2:3]
+; GFX1250-NEXT: s_endpgm
+ %v = load i24, ptr addrspace(4) %in, align 4
+ %c = sitofp i24 %v to float
+ store float %c, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @sitofp_i24_constant_align2(ptr addrspace(4) %in, ptr addrspace(1) %out) {
+; GFX900-LABEL: sitofp_i24_constant_align2:
+; GFX900: ; %bb.0:
+; GFX900-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX900-NEXT: v_mov_b32_e32 v0, 0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: global_load_sbyte v1, v0, s[0:1] offset:2
+; GFX900-NEXT: global_load_ushort v2, v0, s[0:1]
+; GFX900-NEXT: s_waitcnt vmcnt(1)
+; GFX900-NEXT: v_lshlrev_b32_e32 v1, 16, v1
+; GFX900-NEXT: s_waitcnt vmcnt(0)
+; GFX900-NEXT: v_or_b32_e32 v1, v2, v1
+; GFX900-NEXT: v_cvt_f32_i32_e32 v1, v1
+; GFX900-NEXT: global_store_dword v0, v1, s[2:3]
+; GFX900-NEXT: s_endpgm
+;
+; GFX1250-LABEL: sitofp_i24_constant_align2:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT: s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_clause 0x1
+; GFX1250-NEXT: s_load_i8 s4, s[0:1], 0x2 nv
+; GFX1250-NEXT: s_load_u16 s5, s[0:1], 0x0 nv
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_lshl_b32 s0, s4, 16
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_or_b32 s0, s5, s0
+; GFX1250-NEXT: s_cvt_f32_i32 s0, s0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1250-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s0
+; GFX1250-NEXT: global_store_b32 v0, v1, s[2:3]
+; GFX1250-NEXT: s_endpgm
+ %v = load i24, ptr addrspace(4) %in, align 2
+ %c = sitofp i24 %v to float
+ store float %c, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @sitofp_i24_global_invariant_align4(ptr addrspace(1) %in, ptr addrspace(1) %out) {
+; GFX900-LABEL: sitofp_i24_global_invariant_align4:
+; GFX900: ; %bb.0:
+; GFX900-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX900-NEXT: v_mov_b32_e32 v1, 0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: s_load_dword s0, s[0:1], 0x0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: s_bfe_i32 s0, s0, 0x180000
+; GFX900-NEXT: v_cvt_f32_i32_e32 v0, s0
+; GFX900-NEXT: global_store_dword v1, v0, s[2:3]
+; GFX900-NEXT: s_endpgm
+;
+; GFX1250-LABEL: sitofp_i24_global_invariant_align4:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT: s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_load_b32 s0, s[0:1], 0x0 nv
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_bfe_i32 s0, s0, 0x180000
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
+; GFX1250-NEXT: s_cvt_f32_i32 s0, s0
+; GFX1250-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s0
+; GFX1250-NEXT: global_store_b32 v0, v1, s[2:3]
+; GFX1250-NEXT: s_endpgm
+ %v = load i24, ptr addrspace(1) %in, align 4, !invariant.load !0
+ %c = sitofp i24 %v to float
+ store float %c, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @sitofp_i24_global_align4(ptr addrspace(1) %in, ptr addrspace(1) %out) {
+; GFX900-LABEL: sitofp_i24_global_align4:
+; GFX900: ; %bb.0:
+; GFX900-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX900-NEXT: v_mov_b32_e32 v0, 0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: global_load_sbyte v1, v0, s[0:1] offset:2
+; GFX900-NEXT: global_load_ushort v2, v0, s[0:1]
+; GFX900-NEXT: s_waitcnt vmcnt(1)
+; GFX900-NEXT: v_lshlrev_b32_e32 v1, 16, v1
+; GFX900-NEXT: s_waitcnt vmcnt(0)
+; GFX900-NEXT: v_or_b32_e32 v1, v2, v1
+; GFX900-NEXT: v_cvt_f32_i32_e32 v1, v1
+; GFX900-NEXT: global_store_dword v0, v1, s[2:3]
+; GFX900-NEXT: s_endpgm
+;
+; GFX1250-LABEL: sitofp_i24_global_align4:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT: s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_clause 0x1
+; GFX1250-NEXT: s_load_i8 s4, s[0:1], 0x2
+; GFX1250-NEXT: s_load_u16 s5, s[0:1], 0x0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_lshl_b32 s0, s4, 16
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_or_b32 s0, s5, s0
+; GFX1250-NEXT: s_cvt_f32_i32 s0, s0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1250-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s0
+; GFX1250-NEXT: global_store_b32 v0, v1, s[2:3]
+; GFX1250-NEXT: s_endpgm
+ %v = load i24, ptr addrspace(1) %in, align 4
+ %c = sitofp i24 %v to float
+ store float %c, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @sitofp_i48_constant_align8(ptr addrspace(4) %in, ptr addrspace(1) %out) {
+; GFX900-LABEL: sitofp_i48_constant_align8:
+; GFX900: ; %bb.0:
+; GFX900-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX900-NEXT: v_mov_b32_e32 v1, 0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: s_load_dwordx2 s[0:1], s[0:1], 0x0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: s_mov_b32 s4, s1
+; GFX900-NEXT: s_bfe_i64 s[4:5], s[4:5], 0x100000
+; GFX900-NEXT: s_xor_b32 s5, s0, s4
+; GFX900-NEXT: s_mov_b32 s1, s4
+; GFX900-NEXT: s_flbit_i32 s4, s4
+; GFX900-NEXT: s_ashr_i32 s5, s5, 31
+; GFX900-NEXT: s_add_i32 s5, s5, 32
+; GFX900-NEXT: s_add_i32 s4, s4, -1
+; GFX900-NEXT: s_min_u32 s4, s4, s5
+; GFX900-NEXT: s_lshl_b64 s[0:1], s[0:1], s4
+; GFX900-NEXT: s_min_u32 s0, s0, 1
+; GFX900-NEXT: s_or_b32 s0, s1, s0
+; GFX900-NEXT: v_cvt_f32_i32_e32 v0, s0
+; GFX900-NEXT: s_sub_i32 s0, 32, s4
+; GFX900-NEXT: v_ldexp_f32 v0, v0, s0
+; GFX900-NEXT: global_store_dword v1, v0, s[2:3]
+; GFX900-NEXT: s_endpgm
+;
+; GFX1250-LABEL: sitofp_i48_constant_align8:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT: s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT: v_mov_b32_e32 v0, 0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_load_b64 s[0:1], s[0:1], 0x0 nv
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_mov_b32 s4, s1
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_bfe_i64 s[4:5], s[4:5], 0x100000
+; GFX1250-NEXT: s_xor_b32 s1, s0, s4
+; GFX1250-NEXT: s_cls_i32 s5, s4
+; GFX1250-NEXT: s_ashr_i32 s1, s1, 31
+; GFX1250-NEXT: s_add_co_i32 s5, s5, -1
+; GFX1250-NEXT: s_add_co_i32 s6, s1, 32
+; GFX1250-NEXT: s_mov_b32 s1, s4
+; GFX1250-NEXT: s_min_u32 s4, s5, s6
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_lshl_b64 s[0:1], s[0:1], s4
+; GFX1250-NEXT: s_min_u32 s0, s0, 1
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
+; GFX1250-NEXT: s_or_b32 s0, s1, s0
+; GFX1250-NEXT: s_sub_co_i32 s1, 32, s4
+; GFX1250-NEXT: s_cvt_f32_i32 s0, s0
+; GFX1250-NEXT: v_ldexp_f32 v1, s0, s1
+; GFX1250-NEXT: global_store_b32 v0, v1, s[2:3]
+; GFX1250-NEXT: s_endpgm
+ %v = load i48, ptr addrspace(4) %in, align 8
+ %c = sitofp i48 %v to float
+ store float %c, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @uitofp_i48_constant_align8(ptr addrspace(4) %in, ptr addrspace(1) %out) {
+; GFX900-LABEL: uitofp_i48_constant_align8:
+; GFX900: ; %bb.0:
+; GFX900-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX900-NEXT: v_mov_b32_e32 v1, 0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: s_load_dwordx2 s[0:1], s[0:1], 0x0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: s_and_b32 s1, s1, 0xffff
+; GFX900-NEXT: s_flbit_i32_b32 s4, s1
+; GFX900-NEXT: s_min_u32 s4, s4, 32
+; GFX900-NEXT: s_lshl_b64 s[0:1], s[0:1], s4
+; GFX900-NEXT: s_min_u32 s0, s0, 1
+; GFX900-NEXT: s_or_b32 s0, s1, s0
+; GFX900-NEXT: v_cvt_f32_u32_e32 v0, s0
+; GFX900-NEXT: s_sub_i32 s0, 32, s4
+; GFX900-NEXT: v_ldexp_f32 v0, v0, s0
+; GFX900-NEXT: global_store_dword v1, v0, s[2:3]
+; GFX900-NEXT: s_endpgm
+;
+; GFX1250-LABEL: uitofp_i48_constant_align8:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT: s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT: v_mov_b32_e32 v0, 0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_load_b64 s[0:1], s[0:1], 0x0 nv
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_and_b32 s1, s1, 0xffff
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_clz_i32_u32 s4, s1
+; GFX1250-NEXT: s_min_u32 s4, s4, 32
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_lshl_b64 s[0:1], s[0:1], s4
+; GFX1250-NEXT: s_min_u32 s0, s0, 1
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
+; GFX1250-NEXT: s_or_b32 s0, s1, s0
+; GFX1250-NEXT: s_sub_co_i32 s1, 32, s4
+; GFX1250-NEXT: s_cvt_f32_u32 s0, s0
+; GFX1250-NEXT: v_ldexp_f32 v1, s0, s1
+; GFX1250-NEXT: global_store_b32 v0, v1, s[2:3]
+; GFX1250-NEXT: s_endpgm
+ %v = load i48, ptr addrspace(4) %in, align 8
+ %c = uitofp i48 %v to float
+ store float %c, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @sitofp_i48_constant_align2(ptr addrspace(4) %in, ptr addrspace(1) %out) {
+; GFX900-LABEL: sitofp_i48_constant_align2:
+; GFX900: ; %bb.0:
+; GFX900-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX900-NEXT: v_mov_b32_e32 v2, 0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: global_load_sshort v1, v2, s[0:1] offset:4
+; GFX900-NEXT: global_load_dword v0, v2, s[0:1]
+; GFX900-NEXT: s_waitcnt vmcnt(1)
+; GFX900-NEXT: v_ffbh_i32_e32 v4, v1
+; GFX900-NEXT: s_waitcnt vmcnt(0)
+; GFX900-NEXT: v_xor_b32_e32 v3, v0, v1
+; GFX900-NEXT: v_ashrrev_i32_e32 v3, 31, v3
+; GFX900-NEXT: v_add_u32_e32 v3, 32, v3
+; GFX900-NEXT: v_add_u32_e32 v4, -1, v4
+; GFX900-NEXT: v_min_u32_e32 v3, v4, v3
+; GFX900-NEXT: v_lshlrev_b64 v[0:1], v3, v[0:1]
+; GFX900-NEXT: v_min_u32_e32 v0, 1, v0
+; GFX900-NEXT: v_or_b32_e32 v0, v1, v0
+; GFX900-NEXT: v_cvt_f32_i32_e32 v0, v0
+; GFX900-NEXT: v_sub_u32_e32 v1, 32, v3
+; GFX900-NEXT: v_ldexp_f32 v0, v0, v1
+; GFX900-NEXT: global_store_dword v2, v0, s[2:3]
+; GFX900-NEXT: s_endpgm
+;
+; GFX1250-LABEL: sitofp_i48_constant_align2:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT: s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT: v_mov_b32_e32 v2, 0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: global_load_b32 v0, v2, s[0:1] nv
+; GFX1250-NEXT: s_wait_xcnt 0x0
+; GFX1250-NEXT: s_load_i16 s0, s[0:1], 0x4 nv
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: v_mov_b32_e32 v1, s0
+; GFX1250-NEXT: s_wait_loadcnt 0x0
+; GFX1250-NEXT: v_readfirstlane_b32 s1, v0
+; GFX1250-NEXT: s_xor_b32 s1, s1, s0
+; GFX1250-NEXT: s_cls_i32 s0, s0
+; GFX1250-NEXT: s_ashr_i32 s1, s1, 31
+; GFX1250-NEXT: s_add_co_i32 s0, s0, -1
+; GFX1250-NEXT: s_add_co_i32 s1, s1, 32
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_min_u32 s0, s0, s1
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[0:1], s0, v[0:1]
+; GFX1250-NEXT: s_sub_co_i32 s0, 32, s0
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT: v_min_u32_e32 v0, 1, v0
+; GFX1250-NEXT: v_or_b32_e32 v0, v1, v0
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT: v_cvt_f32_i32_e32 v0, v0
+; GFX1250-NEXT: v_ldexp_f32 v0, v0, s0
+; GFX1250-NEXT: global_store_b32 v2, v0, s[2:3]
+; GFX1250-NEXT: s_endpgm
+ %v = load i48, ptr addrspace(4) %in, align 2
+ %c = sitofp i48 %v to float
+ store float %c, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @sitofp_i48_global_invariant_align8(ptr addrspace(1) %in, ptr addrspace(1) %out) {
+; GFX900-LABEL: sitofp_i48_global_invariant_align8:
+; GFX900: ; %bb.0:
+; GFX900-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX900-NEXT: v_mov_b32_e32 v1, 0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: s_load_dwordx2 s[0:1], s[0:1], 0x0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: s_mov_b32 s4, s1
+; GFX900-NEXT: s_bfe_i64 s[4:5], s[4:5], 0x100000
+; GFX900-NEXT: s_xor_b32 s5, s0, s4
+; GFX900-NEXT: s_mov_b32 s1, s4
+; GFX900-NEXT: s_flbit_i32 s4, s4
+; GFX900-NEXT: s_ashr_i32 s5, s5, 31
+; GFX900-NEXT: s_add_i32 s5, s5, 32
+; GFX900-NEXT: s_add_i32 s4, s4, -1
+; GFX900-NEXT: s_min_u32 s4, s4, s5
+; GFX900-NEXT: s_lshl_b64 s[0:1], s[0:1], s4
+; GFX900-NEXT: s_min_u32 s0, s0, 1
+; GFX900-NEXT: s_or_b32 s0, s1, s0
+; GFX900-NEXT: v_cvt_f32_i32_e32 v0, s0
+; GFX900-NEXT: s_sub_i32 s0, 32, s4
+; GFX900-NEXT: v_ldexp_f32 v0, v0, s0
+; GFX900-NEXT: global_store_dword v1, v0, s[2:3]
+; GFX900-NEXT: s_endpgm
+;
+; GFX1250-LABEL: sitofp_i48_global_invariant_align8:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT: s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT: v_mov_b32_e32 v0, 0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_load_b64 s[0:1], s[0:1], 0x0 nv
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_mov_b32 s4, s1
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_bfe_i64 s[4:5], s[4:5], 0x100000
+; GFX1250-NEXT: s_xor_b32 s1, s0, s4
+; GFX1250-NEXT: s_cls_i32 s5, s4
+; GFX1250-NEXT: s_ashr_i32 s1, s1, 31
+; GFX1250-NEXT: s_add_co_i32 s5, s5, -1
+; GFX1250-NEXT: s_add_co_i32 s6, s1, 32
+; GFX1250-NEXT: s_mov_b32 s1, s4
+; GFX1250-NEXT: s_min_u32 s4, s5, s6
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_lshl_b64 s[0:1], s[0:1], s4
+; GFX1250-NEXT: s_min_u32 s0, s0, 1
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
+; GFX1250-NEXT: s_or_b32 s0, s1, s0
+; GFX1250-NEXT: s_sub_co_i32 s1, 32, s4
+; GFX1250-NEXT: s_cvt_f32_i32 s0, s0
+; GFX1250-NEXT: v_ldexp_f32 v1, s0, s1
+; GFX1250-NEXT: global_store_b32 v0, v1, s[2:3]
+; GFX1250-NEXT: s_endpgm
+ %v = load i48, ptr addrspace(1) %in, align 8, !invariant.load !0
+ %c = sitofp i48 %v to float
+ store float %c, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @sitofp_i48_global_align8(ptr addrspace(1) %in, ptr addrspace(1) %out) {
+; GFX900-LABEL: sitofp_i48_global_align8:
+; GFX900: ; %bb.0:
+; GFX900-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX900-NEXT: v_mov_b32_e32 v2, 0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: global_load_sshort v1, v2, s[0:1] offset:4
+; GFX900-NEXT: s_load_dword s0, s[0:1], 0x0
+; GFX900-NEXT: s_waitcnt lgkmcnt(0)
+; GFX900-NEXT: v_mov_b32_e32 v0, s0
+; GFX900-NEXT: s_waitcnt vmcnt(0)
+; GFX900-NEXT: v_xor_b32_e32 v3, s0, v1
+; GFX900-NEXT: v_ffbh_i32_e32 v4, v1
+; GFX900-NEXT: v_ashrrev_i32_e32 v3, 31, v3
+; GFX900-NEXT: v_add_u32_e32 v3, 32, v3
+; GFX900-NEXT: v_add_u32_e32 v4, -1, v4
+; GFX900-NEXT: v_min_u32_e32 v3, v4, v3
+; GFX900-NEXT: v_lshlrev_b64 v[0:1], v3, v[0:1]
+; GFX900-NEXT: v_min_u32_e32 v0, 1, v0
+; GFX900-NEXT: v_or_b32_e32 v0, v1, v0
+; GFX900-NEXT: v_cvt_f32_i32_e32 v0, v0
+; GFX900-NEXT: v_sub_u32_e32 v1, 32, v3
+; GFX900-NEXT: v_ldexp_f32 v0, v0, v1
+; GFX900-NEXT: global_store_dword v2, v0, s[2:3]
+; GFX900-NEXT: s_endpgm
+;
+; GFX1250-LABEL: sitofp_i48_global_align8:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_mov_b64 s[64:65], 0
+; GFX1250-NEXT: v_nop
+; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT: s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT: v_mov_b32_e32 v0, 0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_clause 0x1
+; GFX1250-NEXT: s_load_b32 s4, s[0:1], 0x0
+; GFX1250-NEXT: s_load_i16 s5, s[0:1], 0x4
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_xor_b32 s0, s4, s5
+; GFX1250-NEXT: s_cls_i32 s1, s5
+; GFX1250-NEXT: s_ashr_i32 s0, s0, 31
+; GFX1250-NEXT: s_add_co_i32 s1, s1, -1
+; GFX1250-NEXT: s_add_co_i32 s0, s0, 32
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_min_u32 s6, s1, s0
+; GFX1250-NEXT: s_lshl_b64 s[0:1], s[4:5], s6
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_min_u32 s0, s0, 1
+; GFX1250-NEXT: s_or_b32 s0, s1, s0
+; GFX1250-NEXT: s_sub_co_i32 s1, 32, s6
+; GFX1250-NEXT: s_cvt_f32_i32 s0, s0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1250-NEXT: v_ldexp_f32 v1, s0, s1
+; GFX1250-NEXT: global_store_b32 v0, v1, s[2:3]
+; GFX1250-NEXT: s_endpgm
+ %v = load i48, ptr addrspace(1) %in, align 8
+ %c = sitofp i48 %v to float
+ store float %c, ptr addrspace(1) %out
+ ret void
+}
+
+!0 = !{}
diff --git a/llvm/test/CodeGen/AMDGPU/slp-int-to-fp.ll b/llvm/test/CodeGen/AMDGPU/slp-int-to-fp.ll
index 83d29182d3498..5b73c71c29f13 100644
--- a/llvm/test/CodeGen/AMDGPU/slp-int-to-fp.ll
+++ b/llvm/test/CodeGen/AMDGPU/slp-int-to-fp.ll
@@ -801,3 +801,858 @@ define void @sitofp_i12_to_bfloat(ptr addrspace(1) noalias %out, ptr addrspace(1
store bfloat %c3, ptr addrspace(1) %out.3, align 2
ret void
}
+
+define void @sitofp_i24_to_half(ptr addrspace(1) noalias %out, ptr addrspace(1) noalias %in) {
+; GFX900-LABEL: sitofp_i24_to_half:
+; GFX900: ; %bb.0:
+; GFX900-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX900-NEXT: global_load_ushort v4, v[2:3], off
+; GFX900-NEXT: global_load_sbyte v5, v[2:3], off offset:2
+; GFX900-NEXT: global_load_ushort v6, v[2:3], off offset:4
+; GFX900-NEXT: global_load_sbyte v7, v[2:3], off offset:6
+; GFX900-NEXT: global_load_ushort v8, v[2:3], off offset:8
+; GFX900-NEXT: global_load_sbyte v9, v[2:3], off offset:10
+; GFX900-NEXT: global_load_ushort v10, v[2:3], off offset:12
+; GFX900-NEXT: global_load_sbyte v11, v[2:3], off offset:14
+; GFX900-NEXT: s_waitcnt vmcnt(6)
+; GFX900-NEXT: v_lshl_or_b32 v2, v5, 16, v4
+; GFX900-NEXT: v_cvt_f32_i32_e32 v2, v2
+; GFX900-NEXT: s_waitcnt vmcnt(4)
+; GFX900-NEXT: v_lshl_or_b32 v3, v7, 16, v6
+; GFX900-NEXT: v_cvt_f32_i32_e32 v3, v3
+; GFX900-NEXT: s_waitcnt vmcnt(2)
+; GFX900-NEXT: v_lshl_or_b32 v4, v9, 16, v8
+; GFX900-NEXT: v_cvt_f32_i32_e32 v4, v4
+; GFX900-NEXT: s_waitcnt vmcnt(0)
+; GFX900-NEXT: v_lshl_or_b32 v5, v11, 16, v10
+; GFX900-NEXT: v_cvt_f32_i32_e32 v5, v5
+; GFX900-NEXT: v_cvt_f16_f32_e32 v2, v2
+; GFX900-NEXT: v_cvt_f16_f32_e32 v4, v4
+; GFX900-NEXT: v_cvt_f16_f32_e32 v6, v3
+; GFX900-NEXT: v_cvt_f16_f32_e32 v5, v5
+; GFX900-NEXT: v_pack_b32_f16 v2, v2, v6
+; GFX900-NEXT: v_pack_b32_f16 v3, v4, v5
+; GFX900-NEXT: global_store_dwordx2 v[0:1], v[2:3], off
+; GFX900-NEXT: s_waitcnt vmcnt(0)
+; GFX900-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX90A-LABEL: sitofp_i24_to_half:
+; GFX90A: ; %bb.0:
+; GFX90A-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX90A-NEXT: global_load_ushort v4, v[2:3], off
+; GFX90A-NEXT: global_load_sbyte v5, v[2:3], off offset:2
+; GFX90A-NEXT: global_load_ushort v6, v[2:3], off offset:4
+; GFX90A-NEXT: global_load_sbyte v7, v[2:3], off offset:6
+; GFX90A-NEXT: global_load_ushort v8, v[2:3], off offset:8
+; GFX90A-NEXT: global_load_sbyte v9, v[2:3], off offset:10
+; GFX90A-NEXT: global_load_ushort v10, v[2:3], off offset:12
+; GFX90A-NEXT: global_load_sbyte v11, v[2:3], off offset:14
+; GFX90A-NEXT: s_waitcnt vmcnt(6)
+; GFX90A-NEXT: v_lshl_or_b32 v2, v5, 16, v4
+; GFX90A-NEXT: v_cvt_f32_i32_e32 v2, v2
+; GFX90A-NEXT: s_waitcnt vmcnt(4)
+; GFX90A-NEXT: v_lshl_or_b32 v3, v7, 16, v6
+; GFX90A-NEXT: v_cvt_f32_i32_e32 v3, v3
+; GFX90A-NEXT: s_waitcnt vmcnt(2)
+; GFX90A-NEXT: v_lshl_or_b32 v4, v9, 16, v8
+; GFX90A-NEXT: v_cvt_f32_i32_e32 v4, v4
+; GFX90A-NEXT: s_waitcnt vmcnt(0)
+; GFX90A-NEXT: v_lshl_or_b32 v5, v11, 16, v10
+; GFX90A-NEXT: v_cvt_f32_i32_e32 v5, v5
+; GFX90A-NEXT: v_cvt_f16_f32_e32 v2, v2
+; GFX90A-NEXT: v_cvt_f16_f32_e32 v4, v4
+; GFX90A-NEXT: v_cvt_f16_f32_e32 v6, v3
+; GFX90A-NEXT: v_cvt_f16_f32_e32 v5, v5
+; GFX90A-NEXT: v_pack_b32_f16 v2, v2, v6
+; GFX90A-NEXT: v_pack_b32_f16 v3, v4, v5
+; GFX90A-NEXT: global_store_dwordx2 v[0:1], v[2:3], off
+; GFX90A-NEXT: s_waitcnt vmcnt(0)
+; GFX90A-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX1100-LABEL: sitofp_i24_to_half:
+; GFX1100: ; %bb.0:
+; GFX1100-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1100-NEXT: s_clause 0x7
+; GFX1100-NEXT: global_load_u16 v4, v[2:3], off
+; GFX1100-NEXT: global_load_i8 v5, v[2:3], off offset:2
+; GFX1100-NEXT: global_load_u16 v6, v[2:3], off offset:4
+; GFX1100-NEXT: global_load_i8 v7, v[2:3], off offset:6
+; GFX1100-NEXT: global_load_u16 v8, v[2:3], off offset:8
+; GFX1100-NEXT: global_load_i8 v9, v[2:3], off offset:10
+; GFX1100-NEXT: global_load_u16 v10, v[2:3], off offset:12
+; GFX1100-NEXT: global_load_i8 v2, v[2:3], off offset:14
+; GFX1100-NEXT: s_waitcnt vmcnt(6)
+; GFX1100-NEXT: v_lshl_or_b32 v3, v5, 16, v4
+; GFX1100-NEXT: s_waitcnt vmcnt(4)
+; GFX1100-NEXT: v_lshl_or_b32 v4, v7, 16, v6
+; GFX1100-NEXT: s_waitcnt vmcnt(2)
+; GFX1100-NEXT: v_lshl_or_b32 v5, v9, 16, v8
+; GFX1100-NEXT: v_cvt_f32_i32_e32 v3, v3
+; GFX1100-NEXT: s_waitcnt vmcnt(0)
+; GFX1100-NEXT: v_lshl_or_b32 v2, v2, 16, v10
+; GFX1100-NEXT: v_cvt_f32_i32_e32 v4, v4
+; GFX1100-NEXT: v_cvt_f32_i32_e32 v5, v5
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_cvt_f32_i32_e32 v6, v2
+; GFX1100-NEXT: v_cvt_f16_f32_e32 v2.l, v3
+; GFX1100-NEXT: v_cvt_f16_f32_e32 v2.h, v4
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_cvt_f16_f32_e32 v3.l, v5
+; GFX1100-NEXT: v_cvt_f16_f32_e32 v3.h, v6
+; GFX1100-NEXT: global_store_b64 v[0:1], v[2:3], off
+; GFX1100-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: sitofp_i24_to_half:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_clause 0x7
+; GFX1250-NEXT: global_load_u16 v4, v[2:3], off
+; GFX1250-NEXT: global_load_i8 v5, v[2:3], off offset:2
+; GFX1250-NEXT: global_load_u16 v6, v[2:3], off offset:4
+; GFX1250-NEXT: global_load_u16 v7, v[2:3], off offset:8
+; GFX1250-NEXT: global_load_i8 v8, v[2:3], off offset:10
+; GFX1250-NEXT: global_load_u16 v9, v[2:3], off offset:12
+; GFX1250-NEXT: global_load_i8 v10, v[2:3], off offset:14
+; GFX1250-NEXT: global_load_i8 v11, v[2:3], off offset:6
+; GFX1250-NEXT: s_wait_loadcnt 0x6
+; GFX1250-NEXT: s_wait_xcnt 0x0
+; GFX1250-NEXT: v_lshl_or_b32 v2, v5, 16, v4
+; GFX1250-NEXT: s_wait_loadcnt 0x3
+; GFX1250-NEXT: v_lshl_or_b32 v3, v8, 16, v7
+; GFX1250-NEXT: s_wait_loadcnt 0x1
+; GFX1250-NEXT: v_lshl_or_b32 v4, v10, 16, v9
+; GFX1250-NEXT: s_wait_loadcnt 0x0
+; GFX1250-NEXT: v_lshl_or_b32 v5, v11, 16, v6
+; GFX1250-NEXT: v_cvt_f32_i32_e32 v2, v2
+; GFX1250-NEXT: v_cvt_f32_i32_e32 v3, v3
+; GFX1250-NEXT: v_cvt_f32_i32_e32 v4, v4
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT: v_cvt_f32_i32_e32 v5, v5
+; GFX1250-NEXT: v_cvt_pk_f16_f32 v3, v3, v4
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1250-NEXT: v_cvt_pk_f16_f32 v2, v2, v5
+; GFX1250-NEXT: global_store_b64 v[0:1], v[2:3], off
+; GFX1250-NEXT: s_set_pc_i64 s[30:31]
+ %x0 = load i24, ptr addrspace(1) %in, align 4
+ %c0 = sitofp i24 %x0 to half
+ store half %c0, ptr addrspace(1) %out, align 2
+ %in.1 = getelementptr inbounds i24, ptr addrspace(1) %in, i64 1
+ %x1 = load i24, ptr addrspace(1) %in.1, align 4
+ %c1 = sitofp i24 %x1 to half
+ %out.1 = getelementptr inbounds half, ptr addrspace(1) %out, i64 1
+ store half %c1, ptr addrspace(1) %out.1, align 2
+ %in.2 = getelementptr inbounds i24, ptr addrspace(1) %in, i64 2
+ %x2 = load i24, ptr addrspace(1) %in.2, align 4
+ %c2 = sitofp i24 %x2 to half
+ %out.2 = getelementptr inbounds half, ptr addrspace(1) %out, i64 2
+ store half %c2, ptr addrspace(1) %out.2, align 2
+ %in.3 = getelementptr inbounds i24, ptr addrspace(1) %in, i64 3
+ %x3 = load i24, ptr addrspace(1) %in.3, align 4
+ %c3 = sitofp i24 %x3 to half
+ %out.3 = getelementptr inbounds half, ptr addrspace(1) %out, i64 3
+ store half %c3, ptr addrspace(1) %out.3, align 2
+ ret void
+}
+
+define void @sitofp_i48_to_half(ptr addrspace(1) noalias %out, ptr addrspace(1) noalias %in) {
+; GFX900-LABEL: sitofp_i48_to_half:
+; GFX900: ; %bb.0:
+; GFX900-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX900-NEXT: global_load_ushort v5, v[2:3], off offset:4
+; GFX900-NEXT: global_load_dword v6, v[2:3], off offset:24
+; GFX900-NEXT: global_load_dword v8, v[2:3], off offset:16
+; GFX900-NEXT: global_load_dword v10, v[2:3], off offset:8
+; GFX900-NEXT: global_load_dword v4, v[2:3], off
+; GFX900-NEXT: global_load_ushort v11, v[2:3], off offset:12
+; GFX900-NEXT: global_load_ushort v9, v[2:3], off offset:20
+; GFX900-NEXT: global_load_ushort v7, v[2:3], off offset:28
+; GFX900-NEXT: s_waitcnt vmcnt(3)
+; GFX900-NEXT: v_lshlrev_b64 v[2:3], 16, v[4:5]
+; GFX900-NEXT: s_waitcnt vmcnt(2)
+; GFX900-NEXT: v_lshlrev_b64 v[4:5], 16, v[10:11]
+; GFX900-NEXT: s_waitcnt vmcnt(1)
+; GFX900-NEXT: v_lshlrev_b64 v[8:9], 16, v[8:9]
+; GFX900-NEXT: s_waitcnt vmcnt(0)
+; GFX900-NEXT: v_lshlrev_b64 v[6:7], 16, v[6:7]
+; GFX900-NEXT: v_ashrrev_i64 v[2:3], 16, v[2:3]
+; GFX900-NEXT: v_ashrrev_i64 v[4:5], 16, v[4:5]
+; GFX900-NEXT: v_ashrrev_i64 v[8:9], 16, v[8:9]
+; GFX900-NEXT: v_ashrrev_i64 v[6:7], 16, v[6:7]
+; GFX900-NEXT: v_xor_b32_e32 v10, v4, v5
+; GFX900-NEXT: v_xor_b32_e32 v12, v2, v3
+; GFX900-NEXT: v_xor_b32_e32 v14, v6, v7
+; GFX900-NEXT: v_xor_b32_e32 v16, v8, v9
+; GFX900-NEXT: v_ffbh_i32_e32 v11, v5
+; GFX900-NEXT: v_ffbh_i32_e32 v13, v3
+; GFX900-NEXT: v_ffbh_i32_e32 v15, v7
+; GFX900-NEXT: v_ffbh_i32_e32 v17, v9
+; GFX900-NEXT: v_ashrrev_i32_e32 v10, 31, v10
+; GFX900-NEXT: v_ashrrev_i32_e32 v12, 31, v12
+; GFX900-NEXT: v_ashrrev_i32_e32 v14, 31, v14
+; GFX900-NEXT: v_ashrrev_i32_e32 v16, 31, v16
+; GFX900-NEXT: v_add_u32_e32 v11, -1, v11
+; GFX900-NEXT: v_add_u32_e32 v13, -1, v13
+; GFX900-NEXT: v_add_u32_e32 v15, -1, v15
+; GFX900-NEXT: v_add_u32_e32 v17, -1, v17
+; GFX900-NEXT: v_add_u32_e32 v10, 32, v10
+; GFX900-NEXT: v_add_u32_e32 v12, 32, v12
+; GFX900-NEXT: v_add_u32_e32 v14, 32, v14
+; GFX900-NEXT: v_add_u32_e32 v16, 32, v16
+; GFX900-NEXT: v_min_u32_e32 v10, v11, v10
+; GFX900-NEXT: v_min_u32_e32 v11, v13, v12
+; GFX900-NEXT: v_min_u32_e32 v12, v15, v14
+; GFX900-NEXT: v_min_u32_e32 v13, v17, v16
+; GFX900-NEXT: v_lshlrev_b64 v[4:5], v10, v[4:5]
+; GFX900-NEXT: v_lshlrev_b64 v[2:3], v11, v[2:3]
+; GFX900-NEXT: v_lshlrev_b64 v[6:7], v12, v[6:7]
+; GFX900-NEXT: v_lshlrev_b64 v[8:9], v13, v[8:9]
+; GFX900-NEXT: v_min_u32_e32 v4, 1, v4
+; GFX900-NEXT: v_min_u32_e32 v2, 1, v2
+; GFX900-NEXT: v_min_u32_e32 v6, 1, v6
+; GFX900-NEXT: v_min_u32_e32 v8, 1, v8
+; GFX900-NEXT: v_or_b32_e32 v4, v5, v4
+; GFX900-NEXT: v_or_b32_e32 v2, v3, v2
+; GFX900-NEXT: v_or_b32_e32 v3, v7, v6
+; GFX900-NEXT: v_or_b32_e32 v5, v9, v8
+; GFX900-NEXT: v_cvt_f32_i32_e32 v4, v4
+; GFX900-NEXT: v_cvt_f32_i32_e32 v2, v2
+; GFX900-NEXT: v_cvt_f32_i32_e32 v3, v3
+; GFX900-NEXT: v_cvt_f32_i32_e32 v5, v5
+; GFX900-NEXT: v_sub_u32_e32 v10, 32, v10
+; GFX900-NEXT: v_sub_u32_e32 v11, 32, v11
+; GFX900-NEXT: v_sub_u32_e32 v12, 32, v12
+; GFX900-NEXT: v_sub_u32_e32 v13, 32, v13
+; GFX900-NEXT: v_ldexp_f32 v4, v4, v10
+; GFX900-NEXT: v_ldexp_f32 v2, v2, v11
+; GFX900-NEXT: v_ldexp_f32 v3, v3, v12
+; GFX900-NEXT: v_ldexp_f32 v5, v5, v13
+; GFX900-NEXT: v_cvt_f16_f32_e32 v4, v4
+; GFX900-NEXT: v_cvt_f16_f32_e32 v3, v3
+; GFX900-NEXT: v_cvt_f16_f32_e32 v5, v5
+; GFX900-NEXT: v_cvt_f16_f32_e32 v2, v2
+; GFX900-NEXT: v_pack_b32_f16 v3, v5, v3
+; GFX900-NEXT: v_pack_b32_f16 v2, v2, v4
+; GFX900-NEXT: global_store_dwordx2 v[0:1], v[2:3], off
+; GFX900-NEXT: s_waitcnt vmcnt(0)
+; GFX900-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX90A-LABEL: sitofp_i48_to_half:
+; GFX90A: ; %bb.0:
+; GFX90A-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX90A-NEXT: global_load_ushort v5, v[2:3], off offset:4
+; GFX90A-NEXT: global_load_dword v6, v[2:3], off offset:24
+; GFX90A-NEXT: global_load_dword v8, v[2:3], off offset:16
+; GFX90A-NEXT: global_load_dword v10, v[2:3], off offset:8
+; GFX90A-NEXT: global_load_dword v4, v[2:3], off
+; GFX90A-NEXT: global_load_ushort v11, v[2:3], off offset:12
+; GFX90A-NEXT: global_load_ushort v9, v[2:3], off offset:20
+; GFX90A-NEXT: global_load_ushort v7, v[2:3], off offset:28
+; GFX90A-NEXT: s_waitcnt vmcnt(3)
+; GFX90A-NEXT: v_lshlrev_b64 v[2:3], 16, v[4:5]
+; GFX90A-NEXT: s_waitcnt vmcnt(2)
+; GFX90A-NEXT: v_lshlrev_b64 v[4:5], 16, v[10:11]
+; GFX90A-NEXT: s_waitcnt vmcnt(1)
+; GFX90A-NEXT: v_lshlrev_b64 v[8:9], 16, v[8:9]
+; GFX90A-NEXT: s_waitcnt vmcnt(0)
+; GFX90A-NEXT: v_lshlrev_b64 v[6:7], 16, v[6:7]
+; GFX90A-NEXT: v_ashrrev_i64 v[2:3], 16, v[2:3]
+; GFX90A-NEXT: v_ashrrev_i64 v[4:5], 16, v[4:5]
+; GFX90A-NEXT: v_ashrrev_i64 v[8:9], 16, v[8:9]
+; GFX90A-NEXT: v_ashrrev_i64 v[6:7], 16, v[6:7]
+; GFX90A-NEXT: v_xor_b32_e32 v10, v6, v7
+; GFX90A-NEXT: v_xor_b32_e32 v12, v8, v9
+; GFX90A-NEXT: v_xor_b32_e32 v14, v4, v5
+; GFX90A-NEXT: v_xor_b32_e32 v16, v2, v3
+; GFX90A-NEXT: v_ffbh_i32_e32 v11, v7
+; GFX90A-NEXT: v_ffbh_i32_e32 v13, v9
+; GFX90A-NEXT: v_ffbh_i32_e32 v15, v5
+; GFX90A-NEXT: v_ffbh_i32_e32 v17, v3
+; GFX90A-NEXT: v_ashrrev_i32_e32 v10, 31, v10
+; GFX90A-NEXT: v_ashrrev_i32_e32 v12, 31, v12
+; GFX90A-NEXT: v_ashrrev_i32_e32 v14, 31, v14
+; GFX90A-NEXT: v_ashrrev_i32_e32 v16, 31, v16
+; GFX90A-NEXT: v_add_u32_e32 v11, -1, v11
+; GFX90A-NEXT: v_add_u32_e32 v13, -1, v13
+; GFX90A-NEXT: v_add_u32_e32 v15, -1, v15
+; GFX90A-NEXT: v_add_u32_e32 v17, -1, v17
+; GFX90A-NEXT: v_add_u32_e32 v10, 32, v10
+; GFX90A-NEXT: v_add_u32_e32 v12, 32, v12
+; GFX90A-NEXT: v_add_u32_e32 v14, 32, v14
+; GFX90A-NEXT: v_add_u32_e32 v16, 32, v16
+; GFX90A-NEXT: v_min_u32_e32 v10, v11, v10
+; GFX90A-NEXT: v_min_u32_e32 v11, v13, v12
+; GFX90A-NEXT: v_min_u32_e32 v12, v15, v14
+; GFX90A-NEXT: v_min_u32_e32 v13, v17, v16
+; GFX90A-NEXT: v_lshlrev_b64 v[6:7], v10, v[6:7]
+; GFX90A-NEXT: v_lshlrev_b64 v[8:9], v11, v[8:9]
+; GFX90A-NEXT: v_lshlrev_b64 v[4:5], v12, v[4:5]
+; GFX90A-NEXT: v_lshlrev_b64 v[2:3], v13, v[2:3]
+; GFX90A-NEXT: v_min_u32_e32 v6, 1, v6
+; GFX90A-NEXT: v_min_u32_e32 v8, 1, v8
+; GFX90A-NEXT: v_min_u32_e32 v4, 1, v4
+; GFX90A-NEXT: v_min_u32_e32 v2, 1, v2
+; GFX90A-NEXT: v_or_b32_e32 v6, v7, v6
+; GFX90A-NEXT: v_or_b32_e32 v7, v9, v8
+; GFX90A-NEXT: v_or_b32_e32 v4, v5, v4
+; GFX90A-NEXT: v_or_b32_e32 v2, v3, v2
+; GFX90A-NEXT: v_cvt_f32_i32_e32 v3, v6
+; GFX90A-NEXT: v_cvt_f32_i32_e32 v5, v7
+; GFX90A-NEXT: v_cvt_f32_i32_e32 v4, v4
+; GFX90A-NEXT: v_cvt_f32_i32_e32 v2, v2
+; GFX90A-NEXT: v_sub_u32_e32 v10, 32, v10
+; GFX90A-NEXT: v_sub_u32_e32 v11, 32, v11
+; GFX90A-NEXT: v_sub_u32_e32 v12, 32, v12
+; GFX90A-NEXT: v_sub_u32_e32 v13, 32, v13
+; GFX90A-NEXT: v_ldexp_f32 v3, v3, v10
+; GFX90A-NEXT: v_ldexp_f32 v5, v5, v11
+; GFX90A-NEXT: v_ldexp_f32 v4, v4, v12
+; GFX90A-NEXT: v_ldexp_f32 v2, v2, v13
+; GFX90A-NEXT: v_cvt_f16_f32_e32 v3, v3
+; GFX90A-NEXT: v_cvt_f16_f32_e32 v5, v5
+; GFX90A-NEXT: v_cvt_f16_f32_e32 v4, v4
+; GFX90A-NEXT: v_cvt_f16_f32_e32 v2, v2
+; GFX90A-NEXT: v_pack_b32_f16 v3, v5, v3
+; GFX90A-NEXT: v_pack_b32_f16 v2, v2, v4
+; GFX90A-NEXT: global_store_dwordx2 v[0:1], v[2:3], off
+; GFX90A-NEXT: s_waitcnt vmcnt(0)
+; GFX90A-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX1100-LABEL: sitofp_i48_to_half:
+; GFX1100: ; %bb.0:
+; GFX1100-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1100-NEXT: s_clause 0x7
+; GFX1100-NEXT: global_load_u16 v5, v[2:3], off offset:12
+; GFX1100-NEXT: global_load_b32 v6, v[2:3], off offset:24
+; GFX1100-NEXT: global_load_b32 v8, v[2:3], off offset:16
+; GFX1100-NEXT: global_load_b32 v4, v[2:3], off offset:8
+; GFX1100-NEXT: global_load_u16 v11, v[2:3], off offset:4
+; GFX1100-NEXT: global_load_b32 v10, v[2:3], off
+; GFX1100-NEXT: global_load_u16 v7, v[2:3], off offset:28
+; GFX1100-NEXT: global_load_u16 v9, v[2:3], off offset:20
+; GFX1100-NEXT: s_waitcnt vmcnt(4)
+; GFX1100-NEXT: v_lshlrev_b64 v[2:3], 16, v[4:5]
+; GFX1100-NEXT: s_waitcnt vmcnt(2)
+; GFX1100-NEXT: v_lshlrev_b64 v[4:5], 16, v[10:11]
+; GFX1100-NEXT: s_waitcnt vmcnt(1)
+; GFX1100-NEXT: v_lshlrev_b64 v[6:7], 16, v[6:7]
+; GFX1100-NEXT: s_waitcnt vmcnt(0)
+; GFX1100-NEXT: v_lshlrev_b64 v[8:9], 16, v[8:9]
+; GFX1100-NEXT: v_ashrrev_i64 v[2:3], 16, v[2:3]
+; GFX1100-NEXT: v_ashrrev_i64 v[4:5], 16, v[4:5]
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_ashrrev_i64 v[6:7], 16, v[6:7]
+; GFX1100-NEXT: v_ashrrev_i64 v[8:9], 16, v[8:9]
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4)
+; GFX1100-NEXT: v_xor_b32_e32 v10, v2, v3
+; GFX1100-NEXT: v_cls_i32_e32 v11, v3
+; GFX1100-NEXT: v_xor_b32_e32 v12, v4, v5
+; GFX1100-NEXT: v_xor_b32_e32 v14, v6, v7
+; GFX1100-NEXT: v_xor_b32_e32 v16, v8, v9
+; GFX1100-NEXT: v_cls_i32_e32 v13, v5
+; GFX1100-NEXT: v_cls_i32_e32 v15, v7
+; GFX1100-NEXT: v_cls_i32_e32 v17, v9
+; GFX1100-NEXT: v_ashrrev_i32_e32 v10, 31, v10
+; GFX1100-NEXT: v_ashrrev_i32_e32 v12, 31, v12
+; GFX1100-NEXT: v_ashrrev_i32_e32 v14, 31, v14
+; GFX1100-NEXT: v_ashrrev_i32_e32 v16, 31, v16
+; GFX1100-NEXT: v_add_nc_u32_e32 v11, -1, v11
+; GFX1100-NEXT: v_add_nc_u32_e32 v13, -1, v13
+; GFX1100-NEXT: v_add_nc_u32_e32 v15, -1, v15
+; GFX1100-NEXT: v_add_nc_u32_e32 v17, -1, v17
+; GFX1100-NEXT: v_add_nc_u32_e32 v10, 32, v10
+; GFX1100-NEXT: v_add_nc_u32_e32 v12, 32, v12
+; GFX1100-NEXT: v_add_nc_u32_e32 v14, 32, v14
+; GFX1100-NEXT: v_add_nc_u32_e32 v16, 32, v16
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_min_u32_e32 v10, v11, v10
+; GFX1100-NEXT: v_min_u32_e32 v11, v13, v12
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_min_u32_e32 v12, v15, v14
+; GFX1100-NEXT: v_min_u32_e32 v13, v17, v16
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_lshlrev_b64 v[2:3], v10, v[2:3]
+; GFX1100-NEXT: v_lshlrev_b64 v[4:5], v11, v[4:5]
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_lshlrev_b64 v[6:7], v12, v[6:7]
+; GFX1100-NEXT: v_lshlrev_b64 v[8:9], v13, v[8:9]
+; GFX1100-NEXT: v_sub_nc_u32_e32 v10, 32, v10
+; GFX1100-NEXT: v_sub_nc_u32_e32 v11, 32, v11
+; GFX1100-NEXT: v_min_u32_e32 v2, 1, v2
+; GFX1100-NEXT: v_min_u32_e32 v4, 1, v4
+; GFX1100-NEXT: v_min_u32_e32 v6, 1, v6
+; GFX1100-NEXT: v_min_u32_e32 v8, 1, v8
+; GFX1100-NEXT: v_sub_nc_u32_e32 v12, 32, v12
+; GFX1100-NEXT: v_or_b32_e32 v2, v3, v2
+; GFX1100-NEXT: v_or_b32_e32 v3, v5, v4
+; GFX1100-NEXT: v_or_b32_e32 v4, v7, v6
+; GFX1100-NEXT: v_or_b32_e32 v5, v9, v8
+; GFX1100-NEXT: v_sub_nc_u32_e32 v6, 32, v13
+; GFX1100-NEXT: v_cvt_f32_i32_e32 v2, v2
+; GFX1100-NEXT: v_cvt_f32_i32_e32 v3, v3
+; GFX1100-NEXT: v_cvt_f32_i32_e32 v4, v4
+; GFX1100-NEXT: v_cvt_f32_i32_e32 v5, v5
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_ldexp_f32 v2, v2, v10
+; GFX1100-NEXT: v_ldexp_f32 v3, v3, v11
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_ldexp_f32 v4, v4, v12
+; GFX1100-NEXT: v_ldexp_f32 v5, v5, v6
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_cvt_f16_f32_e32 v2.h, v2
+; GFX1100-NEXT: v_cvt_f16_f32_e32 v2.l, v3
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_cvt_f16_f32_e32 v3.h, v4
+; GFX1100-NEXT: v_cvt_f16_f32_e32 v3.l, v5
+; GFX1100-NEXT: global_store_b64 v[0:1], v[2:3], off
+; GFX1100-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: sitofp_i48_to_half:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_clause 0x7
+; GFX1250-NEXT: global_load_u16 v5, v[2:3], off offset:20
+; GFX1250-NEXT: global_load_u16 v7, v[2:3], off offset:28
+; GFX1250-NEXT: global_load_b32 v6, v[2:3], off offset:24
+; GFX1250-NEXT: global_load_u16 v9, v[2:3], off offset:12
+; GFX1250-NEXT: global_load_b32 v4, v[2:3], off offset:16
+; GFX1250-NEXT: global_load_u16 v11, v[2:3], off offset:4
+; GFX1250-NEXT: global_load_b32 v8, v[2:3], off offset:8
+; GFX1250-NEXT: global_load_b32 v10, v[2:3], off
+; GFX1250-NEXT: s_wait_loadcnt 0x5
+; GFX1250-NEXT: s_wait_xcnt 0x0
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[2:3], 16, v[6:7]
+; GFX1250-NEXT: s_wait_loadcnt 0x3
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[4:5], 16, v[4:5]
+; GFX1250-NEXT: s_wait_loadcnt 0x1
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[6:7], 16, v[8:9]
+; GFX1250-NEXT: s_wait_loadcnt 0x0
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[8:9], 16, v[10:11]
+; GFX1250-NEXT: v_ashrrev_i64 v[2:3], 16, v[2:3]
+; GFX1250-NEXT: v_ashrrev_i64 v[4:5], 16, v[4:5]
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_ashrrev_i64 v[6:7], 16, v[6:7]
+; GFX1250-NEXT: v_ashrrev_i64 v[8:9], 16, v[8:9]
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_cls_i32_e32 v11, v3
+; GFX1250-NEXT: v_xor_b32_e32 v12, v4, v5
+; GFX1250-NEXT: v_xor_b32_e32 v10, v2, v3
+; GFX1250-NEXT: v_cls_i32_e32 v13, v5
+; GFX1250-NEXT: v_xor_b32_e32 v14, v6, v7
+; GFX1250-NEXT: v_dual_add_nc_u32 v11, -1, v11 :: v_dual_bitop2_b32 v16, v8, v9 bitop3:0x14
+; GFX1250-NEXT: v_cls_i32_e32 v15, v7
+; GFX1250-NEXT: v_cls_i32_e32 v17, v9
+; GFX1250-NEXT: v_dual_ashrrev_i32 v12, 31, v12 :: v_dual_ashrrev_i32 v10, 31, v10
+; GFX1250-NEXT: v_dual_add_nc_u32 v13, -1, v13 :: v_dual_ashrrev_i32 v14, 31, v14
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1250-NEXT: v_dual_add_nc_u32 v15, -1, v15 :: v_dual_ashrrev_i32 v16, 31, v16
+; GFX1250-NEXT: v_dual_add_nc_u32 v17, -1, v17 :: v_dual_add_nc_u32 v10, 32, v10
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1250-NEXT: v_dual_add_nc_u32 v12, 32, v12 :: v_dual_add_nc_u32 v14, 32, v14
+; GFX1250-NEXT: v_add_nc_u32_e32 v16, 32, v16
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1250-NEXT: v_min_u32_e32 v10, v11, v10
+; GFX1250-NEXT: v_min_u32_e32 v11, v13, v12
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_min_u32_e32 v12, v15, v14
+; GFX1250-NEXT: v_min_u32_e32 v13, v17, v16
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[2:3], v10, v[2:3]
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[4:5], v11, v[4:5]
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[6:7], v12, v[6:7]
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[8:9], v13, v[8:9]
+; GFX1250-NEXT: v_dual_sub_nc_u32 v10, 32, v10 :: v_dual_sub_nc_u32 v11, 32, v11
+; GFX1250-NEXT: v_min_u32_e32 v2, 1, v2
+; GFX1250-NEXT: v_min_u32_e32 v4, 1, v4
+; GFX1250-NEXT: v_min_u32_e32 v6, 1, v6
+; GFX1250-NEXT: v_min_u32_e32 v8, 1, v8
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_dual_sub_nc_u32 v12, 32, v12 :: v_dual_bitop2_b32 v2, v3, v2 bitop3:0x54
+; GFX1250-NEXT: v_or_b32_e32 v3, v5, v4
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_or_b32_e32 v4, v7, v6
+; GFX1250-NEXT: v_dual_sub_nc_u32 v6, 32, v13 :: v_dual_bitop2_b32 v5, v9, v8 bitop3:0x54
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_cvt_f32_i32_e32 v2, v2
+; GFX1250-NEXT: v_cvt_f32_i32_e32 v3, v3
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_cvt_f32_i32_e32 v4, v4
+; GFX1250-NEXT: v_cvt_f32_i32_e32 v5, v5
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_ldexp_f32 v2, v2, v10
+; GFX1250-NEXT: v_ldexp_f32 v3, v3, v11
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_ldexp_f32 v4, v4, v12
+; GFX1250-NEXT: v_ldexp_f32 v5, v5, v6
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT: v_cvt_pk_f16_f32 v3, v3, v2
+; GFX1250-NEXT: v_cvt_pk_f16_f32 v2, v5, v4
+; GFX1250-NEXT: global_store_b64 v[0:1], v[2:3], off
+; GFX1250-NEXT: s_set_pc_i64 s[30:31]
+ %x0 = load i48, ptr addrspace(1) %in, align 8
+ %c0 = sitofp i48 %x0 to half
+ store half %c0, ptr addrspace(1) %out, align 2
+ %in.1 = getelementptr inbounds i48, ptr addrspace(1) %in, i64 1
+ %x1 = load i48, ptr addrspace(1) %in.1, align 8
+ %c1 = sitofp i48 %x1 to half
+ %out.1 = getelementptr inbounds half, ptr addrspace(1) %out, i64 1
+ store half %c1, ptr addrspace(1) %out.1, align 2
+ %in.2 = getelementptr inbounds i48, ptr addrspace(1) %in, i64 2
+ %x2 = load i48, ptr addrspace(1) %in.2, align 8
+ %c2 = sitofp i48 %x2 to half
+ %out.2 = getelementptr inbounds half, ptr addrspace(1) %out, i64 2
+ store half %c2, ptr addrspace(1) %out.2, align 2
+ %in.3 = getelementptr inbounds i48, ptr addrspace(1) %in, i64 3
+ %x3 = load i48, ptr addrspace(1) %in.3, align 8
+ %c3 = sitofp i48 %x3 to half
+ %out.3 = getelementptr inbounds half, ptr addrspace(1) %out, i64 3
+ store half %c3, ptr addrspace(1) %out.3, align 2
+ ret void
+}
+
+define void @uitofp_i56_to_half(ptr addrspace(1) noalias %out, ptr addrspace(1) noalias %in) {
+; GFX900-LABEL: uitofp_i56_to_half:
+; GFX900: ; %bb.0:
+; GFX900-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX900-NEXT: v_mov_b32_e32 v7, 0
+; GFX900-NEXT: v_mov_b32_e32 v9, 0
+; GFX900-NEXT: global_load_ubyte_d16_hi v7, v[2:3], off offset:6
+; GFX900-NEXT: s_nop 0
+; GFX900-NEXT: global_load_ubyte_d16_hi v9, v[2:3], off offset:14
+; GFX900-NEXT: s_nop 0
+; GFX900-NEXT: global_load_ushort v11, v[2:3], off offset:20
+; GFX900-NEXT: global_load_dword v4, v[2:3], off offset:16
+; GFX900-NEXT: global_load_ushort v12, v[2:3], off offset:12
+; GFX900-NEXT: global_load_dword v6, v[2:3], off offset:8
+; GFX900-NEXT: global_load_ushort v13, v[2:3], off offset:4
+; GFX900-NEXT: v_mov_b32_e32 v5, 0
+; GFX900-NEXT: v_mov_b32_e32 v14, 0
+; GFX900-NEXT: global_load_ubyte_d16_hi v14, v[2:3], off offset:22
+; GFX900-NEXT: s_nop 0
+; GFX900-NEXT: global_load_ubyte_d16_hi v5, v[2:3], off offset:30
+; GFX900-NEXT: s_nop 0
+; GFX900-NEXT: global_load_ushort v15, v[2:3], off offset:28
+; GFX900-NEXT: global_load_dword v8, v[2:3], off
+; GFX900-NEXT: global_load_dword v10, v[2:3], off offset:24
+; GFX900-NEXT: s_waitcnt vmcnt(7)
+; GFX900-NEXT: v_or_b32_e32 v3, v12, v9
+; GFX900-NEXT: s_waitcnt vmcnt(4)
+; GFX900-NEXT: v_or_b32_e32 v11, v11, v14
+; GFX900-NEXT: v_or_b32_e32 v2, v13, v7
+; GFX900-NEXT: s_waitcnt vmcnt(2)
+; GFX900-NEXT: v_or_b32_e32 v12, v15, v5
+; GFX900-NEXT: v_and_b32_e32 v9, 0xffffff, v2
+; GFX900-NEXT: v_and_b32_e32 v7, 0xffffff, v3
+; GFX900-NEXT: v_and_b32_e32 v5, 0xffffff, v11
+; GFX900-NEXT: v_and_b32_e32 v11, 0xffffff, v12
+; GFX900-NEXT: v_ffbh_u32_e32 v2, v7
+; GFX900-NEXT: v_ffbh_u32_e32 v3, v9
+; GFX900-NEXT: v_ffbh_u32_e32 v12, v11
+; GFX900-NEXT: v_ffbh_u32_e32 v13, v5
+; GFX900-NEXT: v_min_u32_e32 v14, 32, v2
+; GFX900-NEXT: v_min_u32_e32 v15, 32, v3
+; GFX900-NEXT: v_min_u32_e32 v12, 32, v12
+; GFX900-NEXT: v_min_u32_e32 v13, 32, v13
+; GFX900-NEXT: v_lshlrev_b64 v[2:3], v14, v[6:7]
+; GFX900-NEXT: s_waitcnt vmcnt(1)
+; GFX900-NEXT: v_lshlrev_b64 v[6:7], v15, v[8:9]
+; GFX900-NEXT: s_waitcnt vmcnt(0)
+; GFX900-NEXT: v_lshlrev_b64 v[8:9], v12, v[10:11]
+; GFX900-NEXT: v_lshlrev_b64 v[4:5], v13, v[4:5]
+; GFX900-NEXT: v_min_u32_e32 v2, 1, v2
+; GFX900-NEXT: v_min_u32_e32 v6, 1, v6
+; GFX900-NEXT: v_min_u32_e32 v8, 1, v8
+; GFX900-NEXT: v_min_u32_e32 v4, 1, v4
+; GFX900-NEXT: v_or_b32_e32 v2, v3, v2
+; GFX900-NEXT: v_or_b32_e32 v3, v7, v6
+; GFX900-NEXT: v_or_b32_e32 v6, v9, v8
+; GFX900-NEXT: v_or_b32_e32 v4, v5, v4
+; GFX900-NEXT: v_cvt_f32_u32_e32 v2, v2
+; GFX900-NEXT: v_cvt_f32_u32_e32 v3, v3
+; GFX900-NEXT: v_cvt_f32_u32_e32 v5, v6
+; GFX900-NEXT: v_cvt_f32_u32_e32 v4, v4
+; GFX900-NEXT: v_sub_u32_e32 v14, 32, v14
+; GFX900-NEXT: v_sub_u32_e32 v15, 32, v15
+; GFX900-NEXT: v_sub_u32_e32 v10, 32, v12
+; GFX900-NEXT: v_sub_u32_e32 v11, 32, v13
+; GFX900-NEXT: v_ldexp_f32 v2, v2, v14
+; GFX900-NEXT: v_ldexp_f32 v3, v3, v15
+; GFX900-NEXT: v_ldexp_f32 v5, v5, v10
+; GFX900-NEXT: v_ldexp_f32 v4, v4, v11
+; GFX900-NEXT: v_cvt_f16_f32_e32 v2, v2
+; GFX900-NEXT: v_cvt_f16_f32_e32 v5, v5
+; GFX900-NEXT: v_cvt_f16_f32_e32 v4, v4
+; GFX900-NEXT: v_cvt_f16_f32_e32 v6, v3
+; GFX900-NEXT: v_pack_b32_f16 v3, v4, v5
+; GFX900-NEXT: v_pack_b32_f16 v2, v6, v2
+; GFX900-NEXT: global_store_dwordx2 v[0:1], v[2:3], off
+; GFX900-NEXT: s_waitcnt vmcnt(0)
+; GFX900-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX90A-LABEL: uitofp_i56_to_half:
+; GFX90A: ; %bb.0:
+; GFX90A-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX90A-NEXT: global_load_ushort v5, v[2:3], off offset:4
+; GFX90A-NEXT: global_load_ubyte v7, v[2:3], off offset:6
+; GFX90A-NEXT: global_load_ushort v9, v[2:3], off offset:12
+; GFX90A-NEXT: global_load_ubyte v11, v[2:3], off offset:14
+; GFX90A-NEXT: global_load_ushort v12, v[2:3], off offset:20
+; GFX90A-NEXT: global_load_ubyte v13, v[2:3], off offset:22
+; GFX90A-NEXT: global_load_ushort v14, v[2:3], off offset:28
+; GFX90A-NEXT: global_load_ubyte v15, v[2:3], off offset:30
+; GFX90A-NEXT: global_load_dword v10, v[2:3], off offset:24
+; GFX90A-NEXT: global_load_dword v8, v[2:3], off offset:16
+; GFX90A-NEXT: global_load_dword v6, v[2:3], off offset:8
+; GFX90A-NEXT: global_load_dword v4, v[2:3], off
+; GFX90A-NEXT: s_waitcnt vmcnt(10)
+; GFX90A-NEXT: v_lshl_or_b32 v2, v7, 16, v5
+; GFX90A-NEXT: v_and_b32_e32 v5, 0xffffff, v2
+; GFX90A-NEXT: s_waitcnt vmcnt(8)
+; GFX90A-NEXT: v_lshl_or_b32 v3, v11, 16, v9
+; GFX90A-NEXT: v_and_b32_e32 v7, 0xffffff, v3
+; GFX90A-NEXT: s_waitcnt vmcnt(6)
+; GFX90A-NEXT: v_lshl_or_b32 v9, v13, 16, v12
+; GFX90A-NEXT: v_and_b32_e32 v9, 0xffffff, v9
+; GFX90A-NEXT: s_waitcnt vmcnt(4)
+; GFX90A-NEXT: v_lshl_or_b32 v11, v15, 16, v14
+; GFX90A-NEXT: v_and_b32_e32 v11, 0xffffff, v11
+; GFX90A-NEXT: v_ffbh_u32_e32 v2, v11
+; GFX90A-NEXT: v_ffbh_u32_e32 v3, v9
+; GFX90A-NEXT: v_ffbh_u32_e32 v12, v7
+; GFX90A-NEXT: v_ffbh_u32_e32 v13, v5
+; GFX90A-NEXT: v_min_u32_e32 v14, 32, v2
+; GFX90A-NEXT: v_min_u32_e32 v15, 32, v3
+; GFX90A-NEXT: v_min_u32_e32 v12, 32, v12
+; GFX90A-NEXT: v_min_u32_e32 v13, 32, v13
+; GFX90A-NEXT: s_waitcnt vmcnt(3)
+; GFX90A-NEXT: v_lshlrev_b64 v[2:3], v14, v[10:11]
+; GFX90A-NEXT: s_waitcnt vmcnt(2)
+; GFX90A-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX90A-NEXT: v_swap_b32 v0, v15
+; GFX90A-NEXT: v_lshlrev_b64 v[8:9], v0, v[8:9]
+; GFX90A-NEXT: v_swap_b32 v15, v0
+; GFX90A-NEXT: s_waitcnt vmcnt(1)
+; GFX90A-NEXT: v_lshlrev_b64 v[6:7], v12, v[6:7]
+; GFX90A-NEXT: s_waitcnt vmcnt(0)
+; GFX90A-NEXT: v_lshlrev_b64 v[4:5], v13, v[4:5]
+; GFX90A-NEXT: v_min_u32_e32 v2, 1, v2
+; GFX90A-NEXT: v_min_u32_e32 v8, 1, v8
+; GFX90A-NEXT: v_min_u32_e32 v6, 1, v6
+; GFX90A-NEXT: v_min_u32_e32 v4, 1, v4
+; GFX90A-NEXT: v_or_b32_e32 v2, v3, v2
+; GFX90A-NEXT: v_or_b32_e32 v3, v9, v8
+; GFX90A-NEXT: v_or_b32_e32 v6, v7, v6
+; GFX90A-NEXT: v_or_b32_e32 v4, v5, v4
+; GFX90A-NEXT: v_cvt_f32_u32_e32 v2, v2
+; GFX90A-NEXT: v_cvt_f32_u32_e32 v3, v3
+; GFX90A-NEXT: v_cvt_f32_u32_e32 v5, v6
+; GFX90A-NEXT: v_cvt_f32_u32_e32 v4, v4
+; GFX90A-NEXT: v_sub_u32_e32 v10, 32, v14
+; GFX90A-NEXT: v_sub_u32_e32 v11, 32, v15
+; GFX90A-NEXT: v_sub_u32_e32 v12, 32, v12
+; GFX90A-NEXT: v_sub_u32_e32 v13, 32, v13
+; GFX90A-NEXT: v_ldexp_f32 v2, v2, v10
+; GFX90A-NEXT: v_ldexp_f32 v3, v3, v11
+; GFX90A-NEXT: v_ldexp_f32 v5, v5, v12
+; GFX90A-NEXT: v_ldexp_f32 v4, v4, v13
+; GFX90A-NEXT: v_cvt_f16_f32_e32 v2, v2
+; GFX90A-NEXT: v_cvt_f16_f32_e32 v3, v3
+; GFX90A-NEXT: v_cvt_f16_f32_e32 v5, v5
+; GFX90A-NEXT: v_cvt_f16_f32_e32 v4, v4
+; GFX90A-NEXT: v_pack_b32_f16 v3, v3, v2
+; GFX90A-NEXT: v_pack_b32_f16 v2, v4, v5
+; GFX90A-NEXT: global_store_dwordx2 v[0:1], v[2:3], off
+; GFX90A-NEXT: s_waitcnt vmcnt(0)
+; GFX90A-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX1100-LABEL: uitofp_i56_to_half:
+; GFX1100: ; %bb.0:
+; GFX1100-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1100-NEXT: v_mov_b16_e32 v5.l, 0
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1100-NEXT: v_mov_b32_e32 v7, v5
+; GFX1100-NEXT: v_mov_b32_e32 v9, v5
+; GFX1100-NEXT: v_mov_b32_e32 v10, v5
+; GFX1100-NEXT: s_clause 0xb
+; GFX1100-NEXT: global_load_d16_hi_u8 v5, v[2:3], off offset:30
+; GFX1100-NEXT: global_load_u16 v11, v[2:3], off offset:28
+; GFX1100-NEXT: global_load_d16_hi_u8 v7, v[2:3], off offset:6
+; GFX1100-NEXT: global_load_d16_hi_u8 v9, v[2:3], off offset:14
+; GFX1100-NEXT: global_load_u16 v12, v[2:3], off offset:20
+; GFX1100-NEXT: global_load_b32 v4, v[2:3], off offset:16
+; GFX1100-NEXT: global_load_u16 v13, v[2:3], off offset:12
+; GFX1100-NEXT: global_load_b32 v6, v[2:3], off offset:8
+; GFX1100-NEXT: global_load_u16 v14, v[2:3], off offset:4
+; GFX1100-NEXT: global_load_d16_hi_u8 v10, v[2:3], off offset:22
+; GFX1100-NEXT: global_load_b32 v8, v[2:3], off offset:24
+; GFX1100-NEXT: global_load_b32 v2, v[2:3], off
+; GFX1100-NEXT: s_waitcnt vmcnt(10)
+; GFX1100-NEXT: v_or_b32_e32 v3, v11, v5
+; GFX1100-NEXT: s_waitcnt vmcnt(5)
+; GFX1100-NEXT: v_or_b32_e32 v5, v13, v9
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1100-NEXT: v_and_b32_e32 v9, 0xffffff, v3
+; GFX1100-NEXT: s_waitcnt vmcnt(3)
+; GFX1100-NEXT: v_or_b32_e32 v11, v14, v7
+; GFX1100-NEXT: s_waitcnt vmcnt(2)
+; GFX1100-NEXT: v_or_b32_e32 v10, v12, v10
+; GFX1100-NEXT: v_and_b32_e32 v7, 0xffffff, v5
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1100-NEXT: v_and_b32_e32 v3, 0xffffff, v11
+; GFX1100-NEXT: v_and_b32_e32 v5, 0xffffff, v10
+; GFX1100-NEXT: v_clz_i32_u32_e32 v10, v9
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_clz_i32_u32_e32 v11, v7
+; GFX1100-NEXT: v_clz_i32_u32_e32 v12, v3
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_clz_i32_u32_e32 v13, v5
+; GFX1100-NEXT: v_min_u32_e32 v10, 32, v10
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_min_u32_e32 v11, 32, v11
+; GFX1100-NEXT: v_min_u32_e32 v12, 32, v12
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4)
+; GFX1100-NEXT: v_min_u32_e32 v13, 32, v13
+; GFX1100-NEXT: s_waitcnt vmcnt(1)
+; GFX1100-NEXT: v_lshlrev_b64 v[8:9], v10, v[8:9]
+; GFX1100-NEXT: v_lshlrev_b64 v[6:7], v11, v[6:7]
+; GFX1100-NEXT: v_sub_nc_u32_e32 v10, 32, v10
+; GFX1100-NEXT: s_waitcnt vmcnt(0)
+; GFX1100-NEXT: v_lshlrev_b64 v[2:3], v12, v[2:3]
+; GFX1100-NEXT: v_lshlrev_b64 v[4:5], v13, v[4:5]
+; GFX1100-NEXT: v_sub_nc_u32_e32 v11, 32, v11
+; GFX1100-NEXT: v_min_u32_e32 v8, 1, v8
+; GFX1100-NEXT: v_min_u32_e32 v6, 1, v6
+; GFX1100-NEXT: v_sub_nc_u32_e32 v12, 32, v12
+; GFX1100-NEXT: v_min_u32_e32 v2, 1, v2
+; GFX1100-NEXT: v_min_u32_e32 v4, 1, v4
+; GFX1100-NEXT: v_or_b32_e32 v8, v9, v8
+; GFX1100-NEXT: v_or_b32_e32 v6, v7, v6
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_or_b32_e32 v2, v3, v2
+; GFX1100-NEXT: v_or_b32_e32 v3, v5, v4
+; GFX1100-NEXT: v_sub_nc_u32_e32 v4, 32, v13
+; GFX1100-NEXT: v_cvt_f32_u32_e32 v5, v8
+; GFX1100-NEXT: v_cvt_f32_u32_e32 v6, v6
+; GFX1100-NEXT: v_cvt_f32_u32_e32 v2, v2
+; GFX1100-NEXT: v_cvt_f32_u32_e32 v3, v3
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_ldexp_f32 v5, v5, v10
+; GFX1100-NEXT: v_ldexp_f32 v6, v6, v11
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_ldexp_f32 v7, v2, v12
+; GFX1100-NEXT: v_ldexp_f32 v4, v3, v4
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_cvt_f16_f32_e32 v3.h, v5
+; GFX1100-NEXT: v_cvt_f16_f32_e32 v2.h, v6
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1100-NEXT: v_cvt_f16_f32_e32 v2.l, v7
+; GFX1100-NEXT: v_cvt_f16_f32_e32 v3.l, v4
+; GFX1100-NEXT: global_store_b64 v[0:1], v[2:3], off
+; GFX1100-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: uitofp_i56_to_half:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: s_clause 0xb
+; GFX1250-NEXT: global_load_u8 v5, v[2:3], off offset:30
+; GFX1250-NEXT: global_load_u8 v7, v[2:3], off offset:22
+; GFX1250-NEXT: global_load_u8 v9, v[2:3], off offset:14
+; GFX1250-NEXT: global_load_u8 v11, v[2:3], off offset:6
+; GFX1250-NEXT: global_load_u16 v12, v[2:3], off offset:28
+; GFX1250-NEXT: global_load_u16 v13, v[2:3], off offset:20
+; GFX1250-NEXT: global_load_u16 v14, v[2:3], off offset:12
+; GFX1250-NEXT: global_load_u16 v15, v[2:3], off offset:4
+; GFX1250-NEXT: global_load_b32 v4, v[2:3], off offset:24
+; GFX1250-NEXT: global_load_b32 v6, v[2:3], off offset:16
+; GFX1250-NEXT: global_load_b32 v8, v[2:3], off offset:8
+; GFX1250-NEXT: global_load_b32 v10, v[2:3], off
+; GFX1250-NEXT: s_wait_loadcnt 0xb
+; GFX1250-NEXT: s_wait_xcnt 0x0
+; GFX1250-NEXT: v_mov_b16_e32 v2.l, v5.l
+; GFX1250-NEXT: s_wait_loadcnt 0xa
+; GFX1250-NEXT: v_mov_b16_e32 v3.l, v7.l
+; GFX1250-NEXT: s_wait_loadcnt 0x9
+; GFX1250-NEXT: v_mov_b16_e32 v5.l, v9.l
+; GFX1250-NEXT: s_wait_loadcnt 0x8
+; GFX1250-NEXT: v_mov_b16_e32 v7.l, v11.l
+; GFX1250-NEXT: v_dual_lshlrev_b32 v2, 16, v2 :: v_dual_lshlrev_b32 v3, 16, v3
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1250-NEXT: v_dual_lshlrev_b32 v9, 16, v5 :: v_dual_lshlrev_b32 v11, 16, v7
+; GFX1250-NEXT: s_wait_loadcnt 0x7
+; GFX1250-NEXT: v_bitop3_b32 v5, v12, 0xffffff, v2 bitop3:0xc8
+; GFX1250-NEXT: s_wait_loadcnt 0x6
+; GFX1250-NEXT: v_bitop3_b32 v7, v13, 0xffffff, v3 bitop3:0xc8
+; GFX1250-NEXT: s_wait_loadcnt 0x5
+; GFX1250-NEXT: v_bitop3_b32 v9, v14, 0xffffff, v9 bitop3:0xc8
+; GFX1250-NEXT: s_wait_loadcnt 0x4
+; GFX1250-NEXT: v_bitop3_b32 v11, v15, 0xffffff, v11 bitop3:0xc8
+; GFX1250-NEXT: v_clz_i32_u32_e32 v2, v5
+; GFX1250-NEXT: v_clz_i32_u32_e32 v3, v7
+; GFX1250-NEXT: v_clz_i32_u32_e32 v12, v9
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_clz_i32_u32_e32 v13, v11
+; GFX1250-NEXT: v_min_u32_e32 v14, 32, v2
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_min_u32_e32 v15, 32, v3
+; GFX1250-NEXT: v_min_u32_e32 v12, 32, v12
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4)
+; GFX1250-NEXT: v_min_u32_e32 v13, 32, v13
+; GFX1250-NEXT: s_wait_loadcnt 0x3
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[2:3], v14, v[4:5]
+; GFX1250-NEXT: s_wait_loadcnt 0x2
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[4:5], v15, v[6:7]
+; GFX1250-NEXT: s_wait_loadcnt 0x1
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[6:7], v12, v[8:9]
+; GFX1250-NEXT: s_wait_loadcnt 0x0
+; GFX1250-NEXT: v_lshlrev_b64_e32 v[8:9], v13, v[10:11]
+; GFX1250-NEXT: v_dual_sub_nc_u32 v10, 32, v14 :: v_dual_sub_nc_u32 v11, 32, v15
+; GFX1250-NEXT: v_min_u32_e32 v2, 1, v2
+; GFX1250-NEXT: v_min_u32_e32 v4, 1, v4
+; GFX1250-NEXT: v_min_u32_e32 v6, 1, v6
+; GFX1250-NEXT: v_min_u32_e32 v8, 1, v8
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_dual_sub_nc_u32 v12, 32, v12 :: v_dual_bitop2_b32 v2, v3, v2 bitop3:0x54
+; GFX1250-NEXT: v_or_b32_e32 v3, v5, v4
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_or_b32_e32 v4, v7, v6
+; GFX1250-NEXT: v_dual_sub_nc_u32 v6, 32, v13 :: v_dual_bitop2_b32 v5, v9, v8 bitop3:0x54
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_cvt_f32_u32_e32 v2, v2
+; GFX1250-NEXT: v_cvt_f32_u32_e32 v3, v3
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_cvt_f32_u32_e32 v4, v4
+; GFX1250-NEXT: v_cvt_f32_u32_e32 v5, v5
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_ldexp_f32 v2, v2, v10
+; GFX1250-NEXT: v_ldexp_f32 v3, v3, v11
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: v_ldexp_f32 v4, v4, v12
+; GFX1250-NEXT: v_ldexp_f32 v5, v5, v6
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT: v_cvt_pk_f16_f32 v3, v3, v2
+; GFX1250-NEXT: v_cvt_pk_f16_f32 v2, v5, v4
+; GFX1250-NEXT: global_store_b64 v[0:1], v[2:3], off
+; GFX1250-NEXT: s_set_pc_i64 s[30:31]
+ %x0 = load i56, ptr addrspace(1) %in, align 8
+ %c0 = uitofp i56 %x0 to half
+ store half %c0, ptr addrspace(1) %out, align 2
+ %in.1 = getelementptr inbounds i56, ptr addrspace(1) %in, i64 1
+ %x1 = load i56, ptr addrspace(1) %in.1, align 8
+ %c1 = uitofp i56 %x1 to half
+ %out.1 = getelementptr inbounds half, ptr addrspace(1) %out, i64 1
+ store half %c1, ptr addrspace(1) %out.1, align 2
+ %in.2 = getelementptr inbounds i56, ptr addrspace(1) %in, i64 2
+ %x2 = load i56, ptr addrspace(1) %in.2, align 8
+ %c2 = uitofp i56 %x2 to half
+ %out.2 = getelementptr inbounds half, ptr addrspace(1) %out, i64 2
+ store half %c2, ptr addrspace(1) %out.2, align 2
+ %in.3 = getelementptr inbounds i56, ptr addrspace(1) %in, i64 3
+ %x3 = load i56, ptr addrspace(1) %in.3, align 8
+ %c3 = uitofp i56 %x3 to half
+ %out.3 = getelementptr inbounds half, ptr addrspace(1) %out, i64 3
+ store half %c3, ptr addrspace(1) %out.3, align 2
+ ret void
+}
More information about the llvm-commits
mailing list