[llvm-branch-commits] [llvm] [AMDGPU] Autogenerate checks in simplify-libcalls.ll. NFC (PR #227934)
Harrison Hao via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Wed Sep 30 19:33:40 PDT 2026
https://github.com/harrisonGPU created https://github.com/llvm/llvm-project/pull/227934
<sub>Stack created with <a href="https://github.com/github/gh-stack">GitHub Stacks CLI</a> • <a href="https://gh.io/stacks-feedback">Give Feedback 💬</a></sub>
>From 13e742a189f675e956ec374f0efb34be9b25f667 Mon Sep 17 00:00:00 2001
From: Harrison Hao <tsworld1314 at gmail.com>
Date: Thu, 1 Oct 2026 09:58:52 +0800
Subject: [PATCH] [AMDGPU] Autogenerate checks in simplify-libcalls.ll. NFC
---
llvm/test/CodeGen/AMDGPU/simplify-libcalls.ll | 1962 ++++++++++++++---
1 file changed, 1684 insertions(+), 278 deletions(-)
diff --git a/llvm/test/CodeGen/AMDGPU/simplify-libcalls.ll b/llvm/test/CodeGen/AMDGPU/simplify-libcalls.ll
index a3699102fc096..3ae46dbf63b03 100644
--- a/llvm/test/CodeGen/AMDGPU/simplify-libcalls.ll
+++ b/llvm/test/CodeGen/AMDGPU/simplify-libcalls.ll
@@ -1,17 +1,47 @@
-; RUN: opt -S -O1 -mtriple=amdgpu-- -amdgpu-simplify-libcall < %s | FileCheck -enable-var-scope -check-prefix=GCN -check-prefix=GCN-POSTLINK %s
-; RUN: opt -S -O1 -mtriple=amdgpu-- -amdgpu-simplify-libcall -amdgpu-prelink -amdgpu-enable-ocl-mangling-mismatch-workaround=0 <%s | FileCheck -enable-var-scope -check-prefix=GCN -check-prefix=GCN-PRELINK %s
-; RUN: opt -S -O1 -mtriple=amdgpu-- -amdgpu-use-native -amdgpu-prelink < %s | FileCheck -enable-var-scope -check-prefix=GCN -check-prefix=GCN-NATIVE %s
-; RUN: opt -S -passes='default<O1>' -mtriple=amdgpu-- -amdgpu-simplify-libcall < %s | FileCheck -enable-var-scope -check-prefix=GCN -check-prefix=GCN-POSTLINK %s
-; RUN: opt -S -passes='default<O1>' -mtriple=amdgpu-- -amdgpu-simplify-libcall -amdgpu-prelink -amdgpu-enable-ocl-mangling-mismatch-workaround=0 <%s | FileCheck -enable-var-scope -check-prefix=GCN -check-prefix=GCN-PRELINK %s
-; RUN: opt -S -passes='default<O1>' -mtriple=amdgpu-- -amdgpu-use-native -amdgpu-prelink < %s | FileCheck -enable-var-scope -check-prefix=GCN -check-prefix=GCN-NATIVE %s
-
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_sincos
-; GCN-POSTLINK: call fast float @_Z3sinf(
-; GCN-POSTLINK: call fast float @_Z3cosf(
-; GCN-PRELINK: call fast float @_Z6sincosfPU3AS5f(
-; GCN-NATIVE: call fast float @_Z10native_sinf(
-; GCN-NATIVE: call fast float @_Z10native_cosf(
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -O1 -mtriple=amdgpu-- -amdgpu-simplify-libcall < %s | FileCheck -check-prefix=GCN -check-prefix=GCN-POSTLINK %s
+; RUN: opt -S -O1 -mtriple=amdgpu-- -amdgpu-simplify-libcall -amdgpu-prelink -amdgpu-enable-ocl-mangling-mismatch-workaround=0 <%s | FileCheck -check-prefix=GCN -check-prefix=GCN-PRELINK %s
+; RUN: opt -S -O1 -mtriple=amdgpu-- -amdgpu-use-native -amdgpu-prelink < %s | FileCheck -check-prefix=GCN -check-prefix=GCN-NATIVE %s
+; RUN: opt -S -passes='default<O1>' -mtriple=amdgpu-- -amdgpu-simplify-libcall < %s | FileCheck -check-prefix=GCN -check-prefix=GCN-POSTLINK %s
+; RUN: opt -S -passes='default<O1>' -mtriple=amdgpu-- -amdgpu-simplify-libcall -amdgpu-prelink -amdgpu-enable-ocl-mangling-mismatch-workaround=0 <%s | FileCheck -check-prefix=GCN -check-prefix=GCN-PRELINK %s
+; RUN: opt -S -passes='default<O1>' -mtriple=amdgpu-- -amdgpu-use-native -amdgpu-prelink < %s | FileCheck -check-prefix=GCN -check-prefix=GCN-NATIVE %s
+
define amdgpu_kernel void @test_sincos(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_sincos(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((4, 8)) [[A:%.*]]) local_unnamed_addr {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @_Z3sinf(float [[TMP]])
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL2:%.*]] = tail call fast float @_Z3cosf(float [[TMP]])
+; GCN-POSTLINK-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: store float [[CALL2]], ptr addrspace(1) [[ARRAYIDX3]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_sincos(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((4, 8)) [[A:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[__SINCOS_:%.*]] = alloca float, align 4, addrspace(5)
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = call fast float @_Z6sincosfPU3AS5f(float [[TMP]], ptr addrspace(5) [[__SINCOS_]])
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(5) [[__SINCOS_]], align 4
+; GCN-PRELINK-NEXT: store float [[TMP0]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL2:%.*]] = call fast float @_Z3cosf(float [[TMP]])
+; GCN-PRELINK-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: store float [[TMP1]], ptr addrspace(1) [[ARRAYIDX3]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_sincos(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((4, 8)) [[A:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast float @_Z10native_sinf(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL2:%.*]] = tail call fast float @_Z10native_cosf(float [[TMP]])
+; GCN-NATIVE-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: store float [[CALL2]], ptr addrspace(1) [[ARRAYIDX3]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3sinf(float %tmp)
@@ -26,13 +56,42 @@ declare float @_Z3sinf(float)
declare float @_Z3cosf(float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_sincos_v2
-; GCN-POSTLINK: call fast <2 x float> @_Z3sinDv2_f(
-; GCN-POSTLINK: call fast <2 x float> @_Z3cosDv2_f(
-; GCN-PRELINK: call fast <2 x float> @_Z6sincosDv2_fPU3AS5S_(
-; GCN-NATIVE: call fast <2 x float> @_Z10native_sinDv2_f(
-; GCN-NATIVE: call fast <2 x float> @_Z10native_cosDv2_f(
define amdgpu_kernel void @test_sincos_v2(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_sincos_v2(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((8, 16)) [[A:%.*]]) local_unnamed_addr {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load <2 x float>, ptr addrspace(1) [[A]], align 8
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast <2 x float> @_Z3sinDv2_f(<2 x float> [[TMP]])
+; GCN-POSTLINK-NEXT: store <2 x float> [[CALL]], ptr addrspace(1) [[A]], align 8
+; GCN-POSTLINK-NEXT: [[CALL2:%.*]] = tail call fast <2 x float> @_Z3cosDv2_f(<2 x float> [[TMP]])
+; GCN-POSTLINK-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 8
+; GCN-POSTLINK-NEXT: store <2 x float> [[CALL2]], ptr addrspace(1) [[ARRAYIDX3]], align 8
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_sincos_v2(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((8, 16)) [[A:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[__SINCOS_:%.*]] = alloca <2 x float>, align 8, addrspace(5)
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load <2 x float>, ptr addrspace(1) [[A]], align 8
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = call fast <2 x float> @_Z6sincosDv2_fPU3AS5S_(<2 x float> [[TMP]], ptr addrspace(5) [[__SINCOS_]])
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = load <2 x float>, ptr addrspace(5) [[__SINCOS_]], align 8
+; GCN-PRELINK-NEXT: store <2 x float> [[TMP0]], ptr addrspace(1) [[A]], align 8
+; GCN-PRELINK-NEXT: [[CALL2:%.*]] = call fast <2 x float> @_Z3cosDv2_f(<2 x float> [[TMP]])
+; GCN-PRELINK-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 8
+; GCN-PRELINK-NEXT: store <2 x float> [[TMP1]], ptr addrspace(1) [[ARRAYIDX3]], align 8
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_sincos_v2(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((8, 16)) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load <2 x float>, ptr addrspace(1) [[A]], align 8
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast <2 x float> @_Z10native_sinDv2_f(<2 x float> [[TMP]])
+; GCN-NATIVE-NEXT: store <2 x float> [[CALL]], ptr addrspace(1) [[A]], align 8
+; GCN-NATIVE-NEXT: [[CALL2:%.*]] = tail call fast <2 x float> @_Z10native_cosDv2_f(<2 x float> [[TMP]])
+; GCN-NATIVE-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 8
+; GCN-NATIVE-NEXT: store <2 x float> [[CALL2]], ptr addrspace(1) [[ARRAYIDX3]], align 8
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load <2 x float>, ptr addrspace(1) %a, align 8
%call = call fast <2 x float> @_Z3sinDv2_f(<2 x float> %tmp)
@@ -47,13 +106,47 @@ declare <2 x float> @_Z3sinDv2_f(<2 x float>)
declare <2 x float> @_Z3cosDv2_f(<2 x float>)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_sincos_v3
-; GCN-POSTLINK: call fast <3 x float> @_Z3sinDv3_f(
-; GCN-POSTLINK: call fast <3 x float> @_Z3cosDv3_f(
-; GCN-PRELINK: call fast <3 x float> @_Z6sincosDv3_fPU3AS5S_(
-; GCN-NATIVE: call fast <3 x float> @_Z10native_sinDv3_f(
-; GCN-NATIVE: call fast <3 x float> @_Z10native_cosDv3_f(
define amdgpu_kernel void @test_sincos_v3(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_sincos_v3(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((16, 32)) [[A:%.*]]) local_unnamed_addr {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = load <3 x float>, ptr addrspace(1) [[A]], align 16
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast <3 x float> @_Z3sinDv3_f(<3 x float> [[TMP0]])
+; GCN-POSTLINK-NEXT: [[EXTRACTVEC6:%.*]] = shufflevector <3 x float> [[CALL]], <3 x float> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>
+; GCN-POSTLINK-NEXT: store <4 x float> [[EXTRACTVEC6]], ptr addrspace(1) [[A]], align 16
+; GCN-POSTLINK-NEXT: [[CALL11:%.*]] = tail call fast <3 x float> @_Z3cosDv3_f(<3 x float> [[TMP0]])
+; GCN-POSTLINK-NEXT: [[ARRAYIDX12:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 16
+; GCN-POSTLINK-NEXT: [[EXTRACTVEC13:%.*]] = shufflevector <3 x float> [[CALL11]], <3 x float> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>
+; GCN-POSTLINK-NEXT: store <4 x float> [[EXTRACTVEC13]], ptr addrspace(1) [[ARRAYIDX12]], align 16
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_sincos_v3(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((16, 32)) [[A:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[__SINCOS_:%.*]] = alloca <3 x float>, align 16, addrspace(5)
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = load <3 x float>, ptr addrspace(1) [[A]], align 16
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = call fast <3 x float> @_Z6sincosDv3_fPU3AS5S_(<3 x float> [[TMP0]], ptr addrspace(5) [[__SINCOS_]])
+; GCN-PRELINK-NEXT: [[EXTRACTVEC13:%.*]] = load <4 x float>, ptr addrspace(5) [[__SINCOS_]], align 16
+; GCN-PRELINK-NEXT: [[EXTRACTVEC6:%.*]] = shufflevector <3 x float> [[TMP1]], <3 x float> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>
+; GCN-PRELINK-NEXT: store <4 x float> [[EXTRACTVEC6]], ptr addrspace(1) [[A]], align 16
+; GCN-PRELINK-NEXT: [[CALL11:%.*]] = call fast <3 x float> @_Z3cosDv3_f(<3 x float> [[TMP0]])
+; GCN-PRELINK-NEXT: [[ARRAYIDX12:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 16
+; GCN-PRELINK-NEXT: store <4 x float> [[EXTRACTVEC13]], ptr addrspace(1) [[ARRAYIDX12]], align 16
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_sincos_v3(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((16, 32)) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP0:%.*]] = load <3 x float>, ptr addrspace(1) [[A]], align 16
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast <3 x float> @_Z10native_sinDv3_f(<3 x float> [[TMP0]])
+; GCN-NATIVE-NEXT: [[EXTRACTVEC6:%.*]] = shufflevector <3 x float> [[CALL]], <3 x float> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>
+; GCN-NATIVE-NEXT: store <4 x float> [[EXTRACTVEC6]], ptr addrspace(1) [[A]], align 16
+; GCN-NATIVE-NEXT: [[CALL11:%.*]] = tail call fast <3 x float> @_Z10native_cosDv3_f(<3 x float> [[TMP0]])
+; GCN-NATIVE-NEXT: [[ARRAYIDX12:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 16
+; GCN-NATIVE-NEXT: [[EXTRACTVEC13:%.*]] = shufflevector <3 x float> [[CALL11]], <3 x float> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>
+; GCN-NATIVE-NEXT: store <4 x float> [[EXTRACTVEC13]], ptr addrspace(1) [[ARRAYIDX12]], align 16
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%loadVec4 = load <4 x float>, ptr addrspace(1) %a, align 16
%extractVec4 = shufflevector <4 x float> %loadVec4, <4 x float> poison, <3 x i32> <i32 0, i32 1, i32 2>
@@ -71,13 +164,42 @@ declare <3 x float> @_Z3sinDv3_f(<3 x float>)
declare <3 x float> @_Z3cosDv3_f(<3 x float>)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_sincos_v4
-; GCN-POSTLINK: call fast <4 x float> @_Z3sinDv4_f(
-; GCN-POSTLINK: call fast <4 x float> @_Z3cosDv4_f(
-; GCN-PRELINK: call fast <4 x float> @_Z6sincosDv4_fPU3AS5S_(
-; GCN-NATIVE: call fast <4 x float> @_Z10native_sinDv4_f(
-; GCN-NATIVE: call fast <4 x float> @_Z10native_cosDv4_f(
define amdgpu_kernel void @test_sincos_v4(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_sincos_v4(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((16, 32)) [[A:%.*]]) local_unnamed_addr {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load <4 x float>, ptr addrspace(1) [[A]], align 16
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast <4 x float> @_Z3sinDv4_f(<4 x float> [[TMP]])
+; GCN-POSTLINK-NEXT: store <4 x float> [[CALL]], ptr addrspace(1) [[A]], align 16
+; GCN-POSTLINK-NEXT: [[CALL2:%.*]] = tail call fast <4 x float> @_Z3cosDv4_f(<4 x float> [[TMP]])
+; GCN-POSTLINK-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 16
+; GCN-POSTLINK-NEXT: store <4 x float> [[CALL2]], ptr addrspace(1) [[ARRAYIDX3]], align 16
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_sincos_v4(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((16, 32)) [[A:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[__SINCOS_:%.*]] = alloca <4 x float>, align 16, addrspace(5)
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load <4 x float>, ptr addrspace(1) [[A]], align 16
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = call fast <4 x float> @_Z6sincosDv4_fPU3AS5S_(<4 x float> [[TMP]], ptr addrspace(5) [[__SINCOS_]])
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr addrspace(5) [[__SINCOS_]], align 16
+; GCN-PRELINK-NEXT: store <4 x float> [[TMP0]], ptr addrspace(1) [[A]], align 16
+; GCN-PRELINK-NEXT: [[CALL2:%.*]] = call fast <4 x float> @_Z3cosDv4_f(<4 x float> [[TMP]])
+; GCN-PRELINK-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 16
+; GCN-PRELINK-NEXT: store <4 x float> [[TMP1]], ptr addrspace(1) [[ARRAYIDX3]], align 16
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_sincos_v4(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((16, 32)) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load <4 x float>, ptr addrspace(1) [[A]], align 16
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast <4 x float> @_Z10native_sinDv4_f(<4 x float> [[TMP]])
+; GCN-NATIVE-NEXT: store <4 x float> [[CALL]], ptr addrspace(1) [[A]], align 16
+; GCN-NATIVE-NEXT: [[CALL2:%.*]] = tail call fast <4 x float> @_Z10native_cosDv4_f(<4 x float> [[TMP]])
+; GCN-NATIVE-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 16
+; GCN-NATIVE-NEXT: store <4 x float> [[CALL2]], ptr addrspace(1) [[ARRAYIDX3]], align 16
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load <4 x float>, ptr addrspace(1) %a, align 16
%call = call fast <4 x float> @_Z3sinDv4_f(<4 x float> %tmp)
@@ -92,13 +214,42 @@ declare <4 x float> @_Z3sinDv4_f(<4 x float>)
declare <4 x float> @_Z3cosDv4_f(<4 x float>)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_sincos_v8
-; GCN-POSTLINK: call fast <8 x float> @_Z3sinDv8_f(
-; GCN-POSTLINK: call fast <8 x float> @_Z3cosDv8_f(
-; GCN-PRELINK: call fast <8 x float> @_Z6sincosDv8_fPU3AS5S_(
-; GCN-NATIVE: call fast <8 x float> @_Z10native_sinDv8_f(
-; GCN-NATIVE: call fast <8 x float> @_Z10native_cosDv8_f(
define amdgpu_kernel void @test_sincos_v8(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_sincos_v8(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((32, 64)) [[A:%.*]]) local_unnamed_addr {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load <8 x float>, ptr addrspace(1) [[A]], align 32
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast <8 x float> @_Z3sinDv8_f(<8 x float> [[TMP]])
+; GCN-POSTLINK-NEXT: store <8 x float> [[CALL]], ptr addrspace(1) [[A]], align 32
+; GCN-POSTLINK-NEXT: [[CALL2:%.*]] = tail call fast <8 x float> @_Z3cosDv8_f(<8 x float> [[TMP]])
+; GCN-POSTLINK-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 32
+; GCN-POSTLINK-NEXT: store <8 x float> [[CALL2]], ptr addrspace(1) [[ARRAYIDX3]], align 32
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_sincos_v8(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((32, 64)) [[A:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[__SINCOS_:%.*]] = alloca <8 x float>, align 32, addrspace(5)
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load <8 x float>, ptr addrspace(1) [[A]], align 32
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = call fast <8 x float> @_Z6sincosDv8_fPU3AS5S_(<8 x float> [[TMP]], ptr addrspace(5) [[__SINCOS_]])
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = load <8 x float>, ptr addrspace(5) [[__SINCOS_]], align 32
+; GCN-PRELINK-NEXT: store <8 x float> [[TMP0]], ptr addrspace(1) [[A]], align 32
+; GCN-PRELINK-NEXT: [[CALL2:%.*]] = call fast <8 x float> @_Z3cosDv8_f(<8 x float> [[TMP]])
+; GCN-PRELINK-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 32
+; GCN-PRELINK-NEXT: store <8 x float> [[TMP1]], ptr addrspace(1) [[ARRAYIDX3]], align 32
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_sincos_v8(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((32, 64)) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load <8 x float>, ptr addrspace(1) [[A]], align 32
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast <8 x float> @_Z10native_sinDv8_f(<8 x float> [[TMP]])
+; GCN-NATIVE-NEXT: store <8 x float> [[CALL]], ptr addrspace(1) [[A]], align 32
+; GCN-NATIVE-NEXT: [[CALL2:%.*]] = tail call fast <8 x float> @_Z10native_cosDv8_f(<8 x float> [[TMP]])
+; GCN-NATIVE-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 32
+; GCN-NATIVE-NEXT: store <8 x float> [[CALL2]], ptr addrspace(1) [[ARRAYIDX3]], align 32
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load <8 x float>, ptr addrspace(1) %a, align 32
%call = call fast <8 x float> @_Z3sinDv8_f(<8 x float> %tmp)
@@ -113,13 +264,42 @@ declare <8 x float> @_Z3sinDv8_f(<8 x float>)
declare <8 x float> @_Z3cosDv8_f(<8 x float>)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_sincos_v16
-; GCN-POSTLINK: call fast <16 x float> @_Z3sinDv16_f(
-; GCN-POSTLINK: call fast <16 x float> @_Z3cosDv16_f(
-; GCN-PRELINK: call fast <16 x float> @_Z6sincosDv16_fPU3AS5S_(
-; GCN-NATIVE: call fast <16 x float> @_Z10native_sinDv16_f(
-; GCN-NATIVE: call fast <16 x float> @_Z10native_cosDv16_f(
define amdgpu_kernel void @test_sincos_v16(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_sincos_v16(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((64, 128)) [[A:%.*]]) local_unnamed_addr {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load <16 x float>, ptr addrspace(1) [[A]], align 64
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast <16 x float> @_Z3sinDv16_f(<16 x float> [[TMP]])
+; GCN-POSTLINK-NEXT: store <16 x float> [[CALL]], ptr addrspace(1) [[A]], align 64
+; GCN-POSTLINK-NEXT: [[CALL2:%.*]] = tail call fast <16 x float> @_Z3cosDv16_f(<16 x float> [[TMP]])
+; GCN-POSTLINK-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 64
+; GCN-POSTLINK-NEXT: store <16 x float> [[CALL2]], ptr addrspace(1) [[ARRAYIDX3]], align 64
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_sincos_v16(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((64, 128)) [[A:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[__SINCOS_:%.*]] = alloca <16 x float>, align 64, addrspace(5)
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load <16 x float>, ptr addrspace(1) [[A]], align 64
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = call fast <16 x float> @_Z6sincosDv16_fPU3AS5S_(<16 x float> [[TMP]], ptr addrspace(5) [[__SINCOS_]])
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = load <16 x float>, ptr addrspace(5) [[__SINCOS_]], align 64
+; GCN-PRELINK-NEXT: store <16 x float> [[TMP0]], ptr addrspace(1) [[A]], align 64
+; GCN-PRELINK-NEXT: [[CALL2:%.*]] = call fast <16 x float> @_Z3cosDv16_f(<16 x float> [[TMP]])
+; GCN-PRELINK-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 64
+; GCN-PRELINK-NEXT: store <16 x float> [[TMP1]], ptr addrspace(1) [[ARRAYIDX3]], align 64
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_sincos_v16(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((64, 128)) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load <16 x float>, ptr addrspace(1) [[A]], align 64
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast <16 x float> @_Z10native_sinDv16_f(<16 x float> [[TMP]])
+; GCN-NATIVE-NEXT: store <16 x float> [[CALL]], ptr addrspace(1) [[A]], align 64
+; GCN-NATIVE-NEXT: [[CALL2:%.*]] = tail call fast <16 x float> @_Z10native_cosDv16_f(<16 x float> [[TMP]])
+; GCN-NATIVE-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 64
+; GCN-NATIVE-NEXT: store <16 x float> [[CALL2]], ptr addrspace(1) [[ARRAYIDX3]], align 64
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load <16 x float>, ptr addrspace(1) %a, align 64
%call = call fast <16 x float> @_Z3sinDv16_f(<16 x float> %tmp)
@@ -134,9 +314,14 @@ declare <16 x float> @_Z3sinDv16_f(<16 x float>)
declare <16 x float> @_Z3cosDv16_f(<16 x float>)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_native_recip
-; GCN: %call = tail call fast float @_Z12native_recipf(float 3.000000e+00)
define amdgpu_kernel void @test_native_recip(ptr addrspace(1) nocapture %a) {
+; GCN-LABEL: define amdgpu_kernel void @test_native_recip(
+; GCN-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr {
+; GCN-NEXT: [[ENTRY:.*:]]
+; GCN-NEXT: [[CALL:%.*]] = tail call fast float @_Z12native_recipf(float 3.000000e+00)
+; GCN-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NEXT: ret void
+;
entry:
%call = call fast float @_Z12native_recipf(float 3.000000e+00)
store float %call, ptr addrspace(1) %a, align 4
@@ -145,9 +330,14 @@ entry:
declare float @_Z12native_recipf(float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_half_recip
-; GCN: %call = tail call fast float @_Z10half_recipf(float 3.000000e+00)
define amdgpu_kernel void @test_half_recip(ptr addrspace(1) nocapture %a) {
+; GCN-LABEL: define amdgpu_kernel void @test_half_recip(
+; GCN-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr {
+; GCN-NEXT: [[ENTRY:.*:]]
+; GCN-NEXT: [[CALL:%.*]] = tail call fast float @_Z10half_recipf(float 3.000000e+00)
+; GCN-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NEXT: ret void
+;
entry:
%call = call fast float @_Z10half_recipf(float 3.000000e+00)
store float %call, ptr addrspace(1) %a, align 4
@@ -158,9 +348,15 @@ declare float @_Z10half_recipf(float)
; Do nothing, the underlying implementation will optimize correctly
; after inlining.
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_native_divide
-; GCN: %call = tail call fast float @_Z13native_divideff(float %tmp, float 3.000000e+00)
define amdgpu_kernel void @test_native_divide(ptr addrspace(1) nocapture %a) {
+; GCN-LABEL: define amdgpu_kernel void @test_native_divide(
+; GCN-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr {
+; GCN-NEXT: [[ENTRY:.*:]]
+; GCN-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NEXT: [[CALL:%.*]] = tail call fast float @_Z13native_divideff(float [[TMP]], float 3.000000e+00)
+; GCN-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z13native_divideff(float %tmp, float 3.000000e+00)
@@ -172,9 +368,15 @@ declare float @_Z13native_divideff(float, float)
; Do nothing, the optimization will naturally happen after inlining.
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_half_divide
-; GCN: %call = tail call fast float @_Z11half_divideff(float %tmp, float 3.000000e+00)
define amdgpu_kernel void @test_half_divide(ptr addrspace(1) nocapture %a) {
+; GCN-LABEL: define amdgpu_kernel void @test_half_divide(
+; GCN-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr {
+; GCN-NEXT: [[ENTRY:.*:]]
+; GCN-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NEXT: [[CALL:%.*]] = tail call fast float @_Z11half_divideff(float [[TMP]], float 3.000000e+00)
+; GCN-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z11half_divideff(float %tmp, float 3.000000e+00)
@@ -184,9 +386,25 @@ entry:
declare float @_Z11half_divideff(float, float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pow_0f
-; GCN: store float 1.000000e+00, ptr addrspace(1) %a
define amdgpu_kernel void @test_pow_0f(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pow_0f(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: store float 1.000000e+00, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pow_0f(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: store float 1.000000e+00, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pow_0f(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1:[0-9]+]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: store float 1.000000e+00, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3powff(float %tmp, float 0.000000e+00)
@@ -196,9 +414,25 @@ entry:
declare float @_Z3powff(float, float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pow_0i
-; GCN: store float 1.000000e+00, ptr addrspace(1) %a
define amdgpu_kernel void @test_pow_0i(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pow_0i(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: store float 1.000000e+00, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pow_0i(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: store float 1.000000e+00, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pow_0i(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: store float 1.000000e+00, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3powff(float %tmp, float 0.000000e+00)
@@ -206,10 +440,31 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pow_1f
-; GCN: %tmp = load float, ptr addrspace(1) %arrayidx, align 4
-; GCN: store float %tmp, ptr addrspace(1) %a, align 4
define amdgpu_kernel void @test_pow_1f(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pow_1f(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1:[0-9]+]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-POSTLINK-NEXT: store float [[TMP]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pow_1f(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1:[0-9]+]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-PRELINK-NEXT: store float [[TMP]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pow_1f(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR2:[0-9]+]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-NATIVE-NEXT: store float [[TMP]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%arrayidx = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
%tmp = load float, ptr addrspace(1) %arrayidx, align 4
@@ -218,10 +473,31 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pow_1i
-; GCN: %tmp = load float, ptr addrspace(1) %arrayidx, align 4
-; GCN: store float %tmp, ptr addrspace(1) %a, align 4
define amdgpu_kernel void @test_pow_1i(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pow_1i(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-POSTLINK-NEXT: store float [[TMP]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pow_1i(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-PRELINK-NEXT: store float [[TMP]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pow_1i(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-NATIVE-NEXT: store float [[TMP]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%arrayidx = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
%tmp = load float, ptr addrspace(1) %arrayidx, align 4
@@ -230,10 +506,31 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pow_2f
-; GCN: %tmp = load float, ptr addrspace(1) %a, align 4
-; GCN: %__pow2 = fmul fast float %tmp, %tmp
define amdgpu_kernel void @test_pow_2f(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pow_2f(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[__POW2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-POSTLINK-NEXT: store float [[__POW2]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pow_2f(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[__POW2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-PRELINK-NEXT: store float [[__POW2]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pow_2f(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[__POW2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-NATIVE-NEXT: store float [[__POW2]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3powff(float %tmp, float 2.000000e+00)
@@ -241,10 +538,31 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pow_2i
-; GCN: %tmp = load float, ptr addrspace(1) %a, align 4
-; GCN: %__pow2 = fmul fast float %tmp, %tmp
define amdgpu_kernel void @test_pow_2i(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pow_2i(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[__POW2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-POSTLINK-NEXT: store float [[__POW2]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pow_2i(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[__POW2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-PRELINK-NEXT: store float [[__POW2]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pow_2i(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[__POW2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-NATIVE-NEXT: store float [[__POW2]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3powff(float %tmp, float 2.000000e+00)
@@ -252,10 +570,34 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pow_m1f
-; GCN: %tmp = load float, ptr addrspace(1) %arrayidx, align 4
-; GCN: %__powrecip = fdiv fast float 1.000000e+00, %tmp
define amdgpu_kernel void @test_pow_m1f(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pow_m1f(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-POSTLINK-NEXT: [[__POWRECIP:%.*]] = fdiv fast float 1.000000e+00, [[TMP]]
+; GCN-POSTLINK-NEXT: store float [[__POWRECIP]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pow_m1f(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-PRELINK-NEXT: [[__POWRECIP:%.*]] = fdiv fast float 1.000000e+00, [[TMP]]
+; GCN-PRELINK-NEXT: store float [[__POWRECIP]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pow_m1f(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-NATIVE-NEXT: [[__POWRECIP:%.*]] = fdiv fast float 1.000000e+00, [[TMP]]
+; GCN-NATIVE-NEXT: store float [[__POWRECIP]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%arrayidx = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
%tmp = load float, ptr addrspace(1) %arrayidx, align 4
@@ -264,10 +606,34 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pow_m1i
-; GCN: %tmp = load float, ptr addrspace(1) %arrayidx, align 4
-; GCN: %__powrecip = fdiv fast float 1.000000e+00, %tmp
define amdgpu_kernel void @test_pow_m1i(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pow_m1i(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-POSTLINK-NEXT: [[__POWRECIP:%.*]] = fdiv fast float 1.000000e+00, [[TMP]]
+; GCN-POSTLINK-NEXT: store float [[__POWRECIP]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pow_m1i(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-PRELINK-NEXT: [[__POWRECIP:%.*]] = fdiv fast float 1.000000e+00, [[TMP]]
+; GCN-PRELINK-NEXT: store float [[__POWRECIP]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pow_m1i(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-NATIVE-NEXT: [[__POWRECIP:%.*]] = fdiv fast float 1.000000e+00, [[TMP]]
+; GCN-NATIVE-NEXT: store float [[__POWRECIP]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%arrayidx = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
%tmp = load float, ptr addrspace(1) %arrayidx, align 4
@@ -276,14 +642,55 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pow_half
-; GCN-POSTLINK: call fast float @llvm.fabs.f32
-; GCN-POSTLINK: call fast float @llvm.log2.f32
-; GCN-POSTLINK: fmul fast float
-; GCN-POSTLINK: call fast float @llvm.exp2.f32
-
-; GCN-PRELINK: %__pow2sqrt = tail call fast float @llvm.sqrt.f32(float %tmp)
define amdgpu_kernel void @test_pow_half(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pow_half(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = fcmp fast oeq float [[TMP]], 1.000000e+00
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = select fast i1 [[TMP0]], float 1.000000e+00, float 5.000000e-01
+; GCN-POSTLINK-NEXT: [[TMP2:%.*]] = tail call fast float @llvm.fabs.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: [[TMP3:%.*]] = tail call fast float @llvm.log2.f32(float [[TMP2]])
+; GCN-POSTLINK-NEXT: [[TMP4:%.*]] = fmul fast float [[TMP1]], [[TMP3]]
+; GCN-POSTLINK-NEXT: [[TMP5:%.*]] = tail call fast float @llvm.exp2.f32(float [[TMP4]])
+; GCN-POSTLINK-NEXT: [[TMP6:%.*]] = tail call fast float @llvm.trunc.f32(float [[TMP1]])
+; GCN-POSTLINK-NEXT: [[TMP7:%.*]] = fcmp fast oeq float [[TMP6]], [[TMP1]]
+; GCN-POSTLINK-NEXT: [[TMP8:%.*]] = fmul fast float [[TMP1]], 5.000000e-01
+; GCN-POSTLINK-NEXT: [[TMP9:%.*]] = tail call fast float @llvm.trunc.f32(float [[TMP8]])
+; GCN-POSTLINK-NEXT: [[TMP10:%.*]] = fcmp fast une float [[TMP9]], [[TMP8]]
+; GCN-POSTLINK-NEXT: [[TMP11:%.*]] = and i1 [[TMP7]], [[TMP10]]
+; GCN-POSTLINK-NEXT: [[TMP12:%.*]] = select fast i1 [[TMP11]], float [[TMP]], float 1.000000e+00
+; GCN-POSTLINK-NEXT: [[TMP13:%.*]] = tail call fast float @llvm.copysign.f32(float [[TMP5]], float [[TMP12]])
+; GCN-POSTLINK-NEXT: [[TMP14:%.*]] = fcmp fast uge float [[TMP]], 0.000000e+00
+; GCN-POSTLINK-NEXT: [[DOTNOT1:%.*]] = or i1 [[TMP14]], [[TMP7]]
+; GCN-POSTLINK-NEXT: [[TMP15:%.*]] = select fast i1 [[DOTNOT1]], float [[TMP13]], float +qnan
+; GCN-POSTLINK-NEXT: [[TMP16:%.*]] = fcmp fast oeq float [[TMP]], 0.000000e+00
+; GCN-POSTLINK-NEXT: [[TMP17:%.*]] = select fast i1 [[TMP16]], float 0.000000e+00, float +inf
+; GCN-POSTLINK-NEXT: [[TMP18:%.*]] = select fast i1 [[TMP11]], float [[TMP]], float 0.000000e+00
+; GCN-POSTLINK-NEXT: [[TMP19:%.*]] = tail call fast float @llvm.copysign.f32(float [[TMP17]], float [[TMP18]])
+; GCN-POSTLINK-NEXT: [[TMP20:%.*]] = select fast i1 [[TMP16]], float [[TMP19]], float [[TMP15]]
+; GCN-POSTLINK-NEXT: store float [[TMP20]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pow_half(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-PRELINK-NEXT: [[__POW2SQRT:%.*]] = tail call fast float @llvm.sqrt.f32(float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[__POW2SQRT]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pow_half(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-NATIVE-NEXT: [[__POW2SQRT:%.*]] = tail call fast float @llvm.sqrt.f32(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[__POW2SQRT]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%arrayidx = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
%tmp = load float, ptr addrspace(1) %arrayidx, align 4
@@ -292,13 +699,57 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pow_mhalf
-; GCN-POSTLINK: call fast float @llvm.fabs.f32
-; GCN-POSTLINK: call fast float @llvm.log2.f32
-; GCN-POSTLINK: fmul fast float
-; GCN-POSTLINK: call fast float @llvm.exp2.f32
-; GCN-PRELINK: %__pow2rsqrt = tail call fast float @_Z5rsqrtf(float %tmp)
define amdgpu_kernel void @test_pow_mhalf(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pow_mhalf(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = fcmp fast oeq float [[TMP]], 1.000000e+00
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = select fast i1 [[TMP0]], float 1.000000e+00, float -5.000000e-01
+; GCN-POSTLINK-NEXT: [[TMP2:%.*]] = tail call fast float @llvm.fabs.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: [[TMP3:%.*]] = tail call fast float @llvm.log2.f32(float [[TMP2]])
+; GCN-POSTLINK-NEXT: [[TMP4:%.*]] = fmul fast float [[TMP1]], [[TMP3]]
+; GCN-POSTLINK-NEXT: [[TMP5:%.*]] = tail call fast float @llvm.exp2.f32(float [[TMP4]])
+; GCN-POSTLINK-NEXT: [[TMP6:%.*]] = tail call fast float @llvm.trunc.f32(float [[TMP1]])
+; GCN-POSTLINK-NEXT: [[TMP7:%.*]] = fcmp fast oeq float [[TMP6]], [[TMP1]]
+; GCN-POSTLINK-NEXT: [[TMP8:%.*]] = fmul fast float [[TMP1]], 5.000000e-01
+; GCN-POSTLINK-NEXT: [[TMP9:%.*]] = tail call fast float @llvm.trunc.f32(float [[TMP8]])
+; GCN-POSTLINK-NEXT: [[TMP10:%.*]] = fcmp fast une float [[TMP9]], [[TMP8]]
+; GCN-POSTLINK-NEXT: [[TMP11:%.*]] = and i1 [[TMP7]], [[TMP10]]
+; GCN-POSTLINK-NEXT: [[TMP12:%.*]] = select fast i1 [[TMP11]], float [[TMP]], float 1.000000e+00
+; GCN-POSTLINK-NEXT: [[TMP13:%.*]] = tail call fast float @llvm.copysign.f32(float [[TMP5]], float [[TMP12]])
+; GCN-POSTLINK-NEXT: [[TMP14:%.*]] = fcmp fast uge float [[TMP]], 0.000000e+00
+; GCN-POSTLINK-NEXT: [[DOTNOT1:%.*]] = or i1 [[TMP14]], [[TMP7]]
+; GCN-POSTLINK-NEXT: [[TMP15:%.*]] = select fast i1 [[DOTNOT1]], float [[TMP13]], float +qnan
+; GCN-POSTLINK-NEXT: [[TMP16:%.*]] = fcmp fast oeq float [[TMP]], 0.000000e+00
+; GCN-POSTLINK-NEXT: [[TMP17:%.*]] = fcmp fast olt float [[TMP1]], 0.000000e+00
+; GCN-POSTLINK-NEXT: [[TMP18:%.*]] = xor i1 [[TMP16]], [[TMP17]]
+; GCN-POSTLINK-NEXT: [[TMP19:%.*]] = select fast i1 [[TMP18]], float 0.000000e+00, float +inf
+; GCN-POSTLINK-NEXT: [[TMP20:%.*]] = select fast i1 [[TMP11]], float [[TMP]], float 0.000000e+00
+; GCN-POSTLINK-NEXT: [[TMP21:%.*]] = tail call fast float @llvm.copysign.f32(float [[TMP19]], float [[TMP20]])
+; GCN-POSTLINK-NEXT: [[TMP22:%.*]] = select fast i1 [[TMP16]], float [[TMP21]], float [[TMP15]]
+; GCN-POSTLINK-NEXT: store float [[TMP22]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pow_mhalf(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-PRELINK-NEXT: [[__POW2RSQRT:%.*]] = tail call fast float @_Z5rsqrtf(float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[__POW2RSQRT]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pow_mhalf(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-NATIVE-NEXT: [[__POW2RSQRT:%.*]] = tail call fast float @_Z12native_rsqrtf(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[__POW2RSQRT]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%arrayidx = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
%tmp = load float, ptr addrspace(1) %arrayidx, align 4
@@ -307,13 +758,46 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pow_c
-; GCN: %__powx2 = fmul fast float %tmp, %tmp
-; GCN: %__powx21 = fmul fast float %__powx2, %__powx2
-; GCN: %__powx22 = fmul fast float %__powx2, %tmp
-; GCN: %[[r0:.*]] = fmul fast float %__powx21, %__powx21
-; GCN: %__powprod3 = fmul fast float %[[r0]], %__powx22
define amdgpu_kernel void @test_pow_c(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pow_c(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-POSTLINK-NEXT: [[__POWX2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-POSTLINK-NEXT: [[__POWX21:%.*]] = fmul fast float [[__POWX2]], [[__POWX2]]
+; GCN-POSTLINK-NEXT: [[__POWX22:%.*]] = fmul fast float [[__POWX2]], [[TMP]]
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = fmul fast float [[__POWX21]], [[__POWX21]]
+; GCN-POSTLINK-NEXT: [[__POWPROD3:%.*]] = fmul fast float [[TMP0]], [[__POWX22]]
+; GCN-POSTLINK-NEXT: store float [[__POWPROD3]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pow_c(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-PRELINK-NEXT: [[__POWX2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-PRELINK-NEXT: [[__POWX21:%.*]] = fmul fast float [[__POWX2]], [[__POWX2]]
+; GCN-PRELINK-NEXT: [[__POWX22:%.*]] = fmul fast float [[__POWX2]], [[TMP]]
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = fmul fast float [[__POWX21]], [[__POWX21]]
+; GCN-PRELINK-NEXT: [[__POWPROD3:%.*]] = fmul fast float [[TMP0]], [[__POWX22]]
+; GCN-PRELINK-NEXT: store float [[__POWPROD3]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pow_c(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-NATIVE-NEXT: [[__POWX2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-NATIVE-NEXT: [[__POWX21:%.*]] = fmul fast float [[__POWX2]], [[__POWX2]]
+; GCN-NATIVE-NEXT: [[__POWX22:%.*]] = fmul fast float [[__POWX2]], [[TMP]]
+; GCN-NATIVE-NEXT: [[TMP0:%.*]] = fmul fast float [[__POWX21]], [[__POWX21]]
+; GCN-NATIVE-NEXT: [[__POWPROD3:%.*]] = fmul fast float [[TMP0]], [[__POWX22]]
+; GCN-NATIVE-NEXT: store float [[__POWPROD3]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%arrayidx = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
%tmp = load float, ptr addrspace(1) %arrayidx, align 4
@@ -322,13 +806,46 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_powr_c
-; GCN: %__powx2 = fmul fast float %tmp, %tmp
-; GCN: %__powx21 = fmul fast float %__powx2, %__powx2
-; GCN: %__powx22 = fmul fast float %__powx2, %tmp
-; GCN: %[[r0:.*]] = fmul fast float %__powx21, %__powx21
-; GCN: %__powprod3 = fmul fast float %[[r0]], %__powx22
define amdgpu_kernel void @test_powr_c(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_powr_c(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-POSTLINK-NEXT: [[__POWX2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-POSTLINK-NEXT: [[__POWX21:%.*]] = fmul fast float [[__POWX2]], [[__POWX2]]
+; GCN-POSTLINK-NEXT: [[__POWX22:%.*]] = fmul fast float [[__POWX2]], [[TMP]]
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = fmul fast float [[__POWX21]], [[__POWX21]]
+; GCN-POSTLINK-NEXT: [[__POWPROD3:%.*]] = fmul fast float [[TMP0]], [[__POWX22]]
+; GCN-POSTLINK-NEXT: store float [[__POWPROD3]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_powr_c(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-PRELINK-NEXT: [[__POWX2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-PRELINK-NEXT: [[__POWX21:%.*]] = fmul fast float [[__POWX2]], [[__POWX2]]
+; GCN-PRELINK-NEXT: [[__POWX22:%.*]] = fmul fast float [[__POWX2]], [[TMP]]
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = fmul fast float [[__POWX21]], [[__POWX21]]
+; GCN-PRELINK-NEXT: [[__POWPROD3:%.*]] = fmul fast float [[TMP0]], [[__POWX22]]
+; GCN-PRELINK-NEXT: store float [[__POWPROD3]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_powr_c(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-NATIVE-NEXT: [[__POWX2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-NATIVE-NEXT: [[__POWX21:%.*]] = fmul fast float [[__POWX2]], [[__POWX2]]
+; GCN-NATIVE-NEXT: [[__POWX22:%.*]] = fmul fast float [[__POWX2]], [[TMP]]
+; GCN-NATIVE-NEXT: [[TMP0:%.*]] = fmul fast float [[__POWX21]], [[__POWX21]]
+; GCN-NATIVE-NEXT: [[__POWPROD3:%.*]] = fmul fast float [[TMP0]], [[__POWX22]]
+; GCN-NATIVE-NEXT: store float [[__POWPROD3]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%arrayidx = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
%tmp = load float, ptr addrspace(1) %arrayidx, align 4
@@ -339,13 +856,46 @@ entry:
declare float @_Z4powrff(float, float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pown_c
-; GCN: %__powx2 = fmul fast float %tmp, %tmp
-; GCN: %__powx21 = fmul fast float %__powx2, %__powx2
-; GCN: %__powx22 = fmul fast float %__powx2, %tmp
-; GCN: %[[r0:.*]] = fmul fast float %__powx21, %__powx21
-; GCN: %__powprod3 = fmul fast float %[[r0]], %__powx22
define amdgpu_kernel void @test_pown_c(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pown_c(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-POSTLINK-NEXT: [[__POWX2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-POSTLINK-NEXT: [[__POWX21:%.*]] = fmul fast float [[__POWX2]], [[__POWX2]]
+; GCN-POSTLINK-NEXT: [[__POWX22:%.*]] = fmul fast float [[__POWX2]], [[TMP]]
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = fmul fast float [[__POWX21]], [[__POWX21]]
+; GCN-POSTLINK-NEXT: [[__POWPROD3:%.*]] = fmul fast float [[TMP0]], [[__POWX22]]
+; GCN-POSTLINK-NEXT: store float [[__POWPROD3]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pown_c(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-PRELINK-NEXT: [[__POWX2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-PRELINK-NEXT: [[__POWX21:%.*]] = fmul fast float [[__POWX2]], [[__POWX2]]
+; GCN-PRELINK-NEXT: [[__POWX22:%.*]] = fmul fast float [[__POWX2]], [[TMP]]
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = fmul fast float [[__POWX21]], [[__POWX21]]
+; GCN-PRELINK-NEXT: [[__POWPROD3:%.*]] = fmul fast float [[TMP0]], [[__POWX22]]
+; GCN-PRELINK-NEXT: store float [[__POWPROD3]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pown_c(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-NATIVE-NEXT: [[__POWX2:%.*]] = fmul fast float [[TMP]], [[TMP]]
+; GCN-NATIVE-NEXT: [[__POWX21:%.*]] = fmul fast float [[__POWX2]], [[__POWX2]]
+; GCN-NATIVE-NEXT: [[__POWX22:%.*]] = fmul fast float [[__POWX2]], [[TMP]]
+; GCN-NATIVE-NEXT: [[TMP0:%.*]] = fmul fast float [[__POWX21]], [[__POWX21]]
+; GCN-NATIVE-NEXT: [[__POWPROD3:%.*]] = fmul fast float [[TMP0]], [[__POWX22]]
+; GCN-NATIVE-NEXT: store float [[__POWPROD3]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%arrayidx = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
%tmp = load float, ptr addrspace(1) %arrayidx, align 4
@@ -356,19 +906,55 @@ entry:
declare half @_Z4pownDhi(half, i32)
-; GCN-LABEL: {{^}}define half @test_pown_f16(
-; GCN-NATIVE: %__fabs = tail call fast half @llvm.fabs.f16(half %x)
-; GCN-NATIVE: %__log2 = tail call fast half @llvm.log2.f16(half %__fabs)
-; GCN-NATIVE: %pownI2F = sitofp fast i32 %y to half
-; GCN-NATIVE: %__ylogx = fmul fast half %__log2, %pownI2F
-; GCN-NATIVE: %__exp2 = tail call fast nofpclass(nan ninf nzero nsub nnorm) half @llvm.exp2.f16(half %__ylogx)
-; GCN-NATIVE: %__ytou = trunc i32 %y to i16
-; GCN-NATIVE: %__yeven = shl i16 %__ytou, 15
-; GCN-NATIVE: %0 = bitcast half %x to i16
-; GCN-NATIVE: %__pow_sign = and i16 %__yeven, %0
-; GCN-NATIVE: %1 = bitcast i16 %__pow_sign to half
-; GCN-NATIVE: %__pow_sign1 = tail call fast half @llvm.copysign.f16(half %__exp2, half %1)
define half @test_pown_f16(half %x, i32 %y) {
+; GCN-POSTLINK-LABEL: define half @test_pown_f16(
+; GCN-POSTLINK-SAME: half [[X:%.*]], i32 [[Y:%.*]]) local_unnamed_addr #[[ATTR2:[0-9]+]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[__FABS:%.*]] = tail call fast half @llvm.fabs.f16(half [[X]])
+; GCN-POSTLINK-NEXT: [[__LOG2:%.*]] = tail call fast half @llvm.log2.f16(half [[__FABS]])
+; GCN-POSTLINK-NEXT: [[POWNI2F:%.*]] = sitofp fast i32 [[Y]] to half
+; GCN-POSTLINK-NEXT: [[__YLOGX:%.*]] = fmul fast half [[__LOG2]], [[POWNI2F]]
+; GCN-POSTLINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) half @llvm.exp2.f16(half [[__YLOGX]])
+; GCN-POSTLINK-NEXT: [[__YTOU:%.*]] = trunc i32 [[Y]] to i16
+; GCN-POSTLINK-NEXT: [[__YEVEN:%.*]] = shl i16 [[__YTOU]], 15
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = bitcast half [[X]] to i16
+; GCN-POSTLINK-NEXT: [[__POW_SIGN:%.*]] = and i16 [[__YEVEN]], [[TMP0]]
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = bitcast i16 [[__POW_SIGN]] to half
+; GCN-POSTLINK-NEXT: [[__POW_SIGN1:%.*]] = tail call fast half @llvm.copysign.f16(half [[__EXP2]], half [[TMP1]])
+; GCN-POSTLINK-NEXT: ret half [[__POW_SIGN1]]
+;
+; GCN-PRELINK-LABEL: define half @test_pown_f16(
+; GCN-PRELINK-SAME: half [[X:%.*]], i32 [[Y:%.*]]) local_unnamed_addr #[[ATTR2:[0-9]+]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[__FABS:%.*]] = tail call fast half @llvm.fabs.f16(half [[X]])
+; GCN-PRELINK-NEXT: [[__LOG2:%.*]] = tail call fast half @llvm.log2.f16(half [[__FABS]])
+; GCN-PRELINK-NEXT: [[POWNI2F:%.*]] = sitofp fast i32 [[Y]] to half
+; GCN-PRELINK-NEXT: [[__YLOGX:%.*]] = fmul fast half [[__LOG2]], [[POWNI2F]]
+; GCN-PRELINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) half @llvm.exp2.f16(half [[__YLOGX]])
+; GCN-PRELINK-NEXT: [[__YTOU:%.*]] = trunc i32 [[Y]] to i16
+; GCN-PRELINK-NEXT: [[__YEVEN:%.*]] = shl i16 [[__YTOU]], 15
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = bitcast half [[X]] to i16
+; GCN-PRELINK-NEXT: [[__POW_SIGN:%.*]] = and i16 [[__YEVEN]], [[TMP0]]
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = bitcast i16 [[__POW_SIGN]] to half
+; GCN-PRELINK-NEXT: [[__POW_SIGN1:%.*]] = tail call fast half @llvm.copysign.f16(half [[__EXP2]], half [[TMP1]])
+; GCN-PRELINK-NEXT: ret half [[__POW_SIGN1]]
+;
+; GCN-NATIVE-LABEL: define half @test_pown_f16(
+; GCN-NATIVE-SAME: half [[X:%.*]], i32 [[Y:%.*]]) local_unnamed_addr #[[ATTR3:[0-9]+]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[__FABS:%.*]] = tail call fast half @llvm.fabs.f16(half [[X]])
+; GCN-NATIVE-NEXT: [[__LOG2:%.*]] = tail call fast half @llvm.log2.f16(half [[__FABS]])
+; GCN-NATIVE-NEXT: [[POWNI2F:%.*]] = sitofp fast i32 [[Y]] to half
+; GCN-NATIVE-NEXT: [[__YLOGX:%.*]] = fmul fast half [[__LOG2]], [[POWNI2F]]
+; GCN-NATIVE-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) half @llvm.exp2.f16(half [[__YLOGX]])
+; GCN-NATIVE-NEXT: [[__YTOU:%.*]] = trunc i32 [[Y]] to i16
+; GCN-NATIVE-NEXT: [[__YEVEN:%.*]] = shl i16 [[__YTOU]], 15
+; GCN-NATIVE-NEXT: [[TMP0:%.*]] = bitcast half [[X]] to i16
+; GCN-NATIVE-NEXT: [[__POW_SIGN:%.*]] = and i16 [[__YEVEN]], [[TMP0]]
+; GCN-NATIVE-NEXT: [[TMP1:%.*]] = bitcast i16 [[__POW_SIGN]] to half
+; GCN-NATIVE-NEXT: [[__POW_SIGN1:%.*]] = tail call fast half @llvm.copysign.f16(half [[__EXP2]], half [[TMP1]])
+; GCN-NATIVE-NEXT: ret half [[__POW_SIGN1]]
+;
entry:
%call = call fast half @_Z4pownDhi(half %x, i32 %y)
ret half %call
@@ -376,14 +962,43 @@ entry:
declare float @_Z4pownfi(float, i32)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pow
-; GCN: %__fabs = tail call fast float @llvm.fabs.f32(float %tmp)
-; GCN: %__log2 = tail call fast float @llvm.log2.f32(float %__fabs)
-; GCN: %__ylogx = fmul fast float %__log2, 1.013000e+03
-; GCN: %__exp2 = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float %__ylogx)
-; GCN: %[[r0:.*]] = tail call fast float @llvm.copysign.f32(float %__exp2, float %tmp)
-; GCN: store float %[[r0]], ptr addrspace(1) %a, align 4
define amdgpu_kernel void @test_pow(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pow(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[__FABS:%.*]] = tail call fast float @llvm.fabs.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: [[__LOG2:%.*]] = tail call fast float @llvm.log2.f32(float [[__FABS]])
+; GCN-POSTLINK-NEXT: [[__YLOGX:%.*]] = fmul fast float [[__LOG2]], 1.013000e+03
+; GCN-POSTLINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float [[__YLOGX]])
+; GCN-POSTLINK-NEXT: [[__POW_SIGN1:%.*]] = tail call fast float @llvm.copysign.f32(float [[__EXP2]], float [[TMP]])
+; GCN-POSTLINK-NEXT: store float [[__POW_SIGN1]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pow(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[__FABS:%.*]] = tail call fast float @llvm.fabs.f32(float [[TMP]])
+; GCN-PRELINK-NEXT: [[__LOG2:%.*]] = tail call fast float @llvm.log2.f32(float [[__FABS]])
+; GCN-PRELINK-NEXT: [[__YLOGX:%.*]] = fmul fast float [[__LOG2]], 1.013000e+03
+; GCN-PRELINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float [[__YLOGX]])
+; GCN-PRELINK-NEXT: [[__POW_SIGN1:%.*]] = tail call fast float @llvm.copysign.f32(float [[__EXP2]], float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[__POW_SIGN1]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pow(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[__FABS:%.*]] = tail call fast float @llvm.fabs.f32(float [[TMP]])
+; GCN-NATIVE-NEXT: [[__LOG2:%.*]] = tail call fast float @llvm.log2.f32(float [[__FABS]])
+; GCN-NATIVE-NEXT: [[__YLOGX:%.*]] = fmul fast float [[__LOG2]], 1.013000e+03
+; GCN-NATIVE-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float [[__YLOGX]])
+; GCN-NATIVE-NEXT: [[__POW_SIGN1:%.*]] = tail call fast float @llvm.copysign.f32(float [[__EXP2]], float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[__POW_SIGN1]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3powff(float %tmp, float 1.013000e+03)
@@ -391,40 +1006,118 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_powr
-; GCN: %__log2 = tail call fast float @llvm.log2.f32(float %tmp)
-; GCN: %__ylogx = fmul fast float %tmp1, %__log2
-; GCN: %__exp2 = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float %__ylogx)
-; GCN: store float %__exp2, ptr addrspace(1) %a, align 4
define amdgpu_kernel void @test_powr(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_powr(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-POSTLINK-NEXT: [[__LOG2:%.*]] = tail call fast float @llvm.log2.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: [[__YLOGX:%.*]] = fmul fast float [[TMP1]], [[__LOG2]]
+; GCN-POSTLINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float [[__YLOGX]])
+; GCN-POSTLINK-NEXT: store float [[__EXP2]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_powr(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-PRELINK-NEXT: [[__LOG2:%.*]] = tail call fast float @llvm.log2.f32(float [[TMP]])
+; GCN-PRELINK-NEXT: [[__YLOGX:%.*]] = fmul fast float [[TMP1]], [[__LOG2]]
+; GCN-PRELINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float [[__YLOGX]])
+; GCN-PRELINK-NEXT: store float [[__EXP2]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_powr(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-NATIVE-NEXT: [[__LOG2:%.*]] = tail call fast float @llvm.log2.f32(float [[TMP]])
+; GCN-NATIVE-NEXT: [[__YLOGX:%.*]] = fmul fast float [[TMP1]], [[__LOG2]]
+; GCN-NATIVE-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float [[__YLOGX]])
+; GCN-NATIVE-NEXT: store float [[__EXP2]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%arrayidx1 = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
- %tmp1 = load float, ptr addrspace(1) %arrayidx1, align 4
- %call = call fast float @_Z4powrff(float %tmp, float %tmp1)
+ %val1 = load float, ptr addrspace(1) %arrayidx1, align 4
+ %call = call fast float @_Z4powrff(float %tmp, float %val1)
store float %call, ptr addrspace(1) %a, align 4
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pown
-; GCN: %conv = fptosi float %tmp1 to i32
-; GCN: %__fabs = tail call fast float @llvm.fabs.f32(float %tmp)
-; GCN: %__log2 = tail call fast float @llvm.log2.f32(float %__fabs)
-; GCN: %pownI2F = sitofp fast i32 %conv to float
-; GCN: %__ylogx = fmul fast float %__log2, %pownI2F
-; GCN: %__exp2 = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float %__ylogx)
-; GCN: %__yeven = shl i32 %conv, 31
-; GCN: %[[r0:.*]] = bitcast float %tmp to i32
-; GCN: %__pow_sign = and i32 %__yeven, %[[r0]]
-; GCN: %[[r1:.*]] = bitcast i32 %__pow_sign to float
-; GCN: %[[r2:.*]] = tail call fast float @llvm.copysign.f32(float %__exp2, float %[[r1]])
-; GCN: store float %[[r2]], ptr addrspace(1) %a, align 4
define amdgpu_kernel void @test_pown(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pown(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-POSTLINK-NEXT: [[CONV:%.*]] = fptosi float [[TMP1]] to i32
+; GCN-POSTLINK-NEXT: [[__FABS:%.*]] = tail call fast float @llvm.fabs.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: [[__LOG2:%.*]] = tail call fast float @llvm.log2.f32(float [[__FABS]])
+; GCN-POSTLINK-NEXT: [[POWNI2F:%.*]] = sitofp fast i32 [[CONV]] to float
+; GCN-POSTLINK-NEXT: [[__YLOGX:%.*]] = fmul fast float [[__LOG2]], [[POWNI2F]]
+; GCN-POSTLINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float [[__YLOGX]])
+; GCN-POSTLINK-NEXT: [[__YEVEN:%.*]] = shl i32 [[CONV]], 31
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = bitcast float [[TMP]] to i32
+; GCN-POSTLINK-NEXT: [[__POW_SIGN:%.*]] = and i32 [[__YEVEN]], [[TMP0]]
+; GCN-POSTLINK-NEXT: [[TMP2:%.*]] = bitcast i32 [[__POW_SIGN]] to float
+; GCN-POSTLINK-NEXT: [[__POW_SIGN1:%.*]] = tail call fast float @llvm.copysign.f32(float [[__EXP2]], float [[TMP2]])
+; GCN-POSTLINK-NEXT: store float [[__POW_SIGN1]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pown(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-PRELINK-NEXT: [[CONV:%.*]] = fptosi float [[TMP1]] to i32
+; GCN-PRELINK-NEXT: [[__FABS:%.*]] = tail call fast float @llvm.fabs.f32(float [[TMP]])
+; GCN-PRELINK-NEXT: [[__LOG2:%.*]] = tail call fast float @llvm.log2.f32(float [[__FABS]])
+; GCN-PRELINK-NEXT: [[POWNI2F:%.*]] = sitofp fast i32 [[CONV]] to float
+; GCN-PRELINK-NEXT: [[__YLOGX:%.*]] = fmul fast float [[__LOG2]], [[POWNI2F]]
+; GCN-PRELINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float [[__YLOGX]])
+; GCN-PRELINK-NEXT: [[__YEVEN:%.*]] = shl i32 [[CONV]], 31
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = bitcast float [[TMP]] to i32
+; GCN-PRELINK-NEXT: [[__POW_SIGN:%.*]] = and i32 [[__YEVEN]], [[TMP0]]
+; GCN-PRELINK-NEXT: [[TMP2:%.*]] = bitcast i32 [[__POW_SIGN]] to float
+; GCN-PRELINK-NEXT: [[__POW_SIGN1:%.*]] = tail call fast float @llvm.copysign.f32(float [[__EXP2]], float [[TMP2]])
+; GCN-PRELINK-NEXT: store float [[__POW_SIGN1]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pown(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-NATIVE-NEXT: [[CONV:%.*]] = fptosi float [[TMP1]] to i32
+; GCN-NATIVE-NEXT: [[__FABS:%.*]] = tail call fast float @llvm.fabs.f32(float [[TMP]])
+; GCN-NATIVE-NEXT: [[__LOG2:%.*]] = tail call fast float @llvm.log2.f32(float [[__FABS]])
+; GCN-NATIVE-NEXT: [[POWNI2F:%.*]] = sitofp fast i32 [[CONV]] to float
+; GCN-NATIVE-NEXT: [[__YLOGX:%.*]] = fmul fast float [[__LOG2]], [[POWNI2F]]
+; GCN-NATIVE-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float [[__YLOGX]])
+; GCN-NATIVE-NEXT: [[__YEVEN:%.*]] = shl i32 [[CONV]], 31
+; GCN-NATIVE-NEXT: [[TMP0:%.*]] = bitcast float [[TMP]] to i32
+; GCN-NATIVE-NEXT: [[__POW_SIGN:%.*]] = and i32 [[__YEVEN]], [[TMP0]]
+; GCN-NATIVE-NEXT: [[TMP2:%.*]] = bitcast i32 [[__POW_SIGN]] to float
+; GCN-NATIVE-NEXT: [[__POW_SIGN1:%.*]] = tail call fast float @llvm.copysign.f32(float [[__EXP2]], float [[TMP2]])
+; GCN-NATIVE-NEXT: store float [[__POW_SIGN1]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%arrayidx1 = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
- %tmp1 = load float, ptr addrspace(1) %arrayidx1, align 4
- %conv = fptosi float %tmp1 to i32
+ %val1 = load float, ptr addrspace(1) %arrayidx1, align 4
+ %conv = fptosi float %val1 to i32
%call = call fast float @_Z4pownfi(float %tmp, i32 %conv)
store float %call, ptr addrspace(1) %a, align 4
ret void
@@ -433,32 +1126,95 @@ entry:
declare half @_Z3powDhDh(half, half)
declare <2 x half> @_Z3powDv2_DhS_(<2 x half>, <2 x half>)
-; GCN-LABEL: define half @test_pow_fast_f16__y_13(half %x)
-; GCN: %__fabs = tail call fast half @llvm.fabs.f16(half %x)
-; GCN: %__log2 = tail call fast half @llvm.log2.f16(half %__fabs)
-; GCN: %__ylogx = fmul fast half %__log2, 1.300000e+01
-; GCN: %__exp2 = tail call fast nofpclass(nan ninf nzero nsub nnorm) half @llvm.exp2.f16(half %__ylogx)
-; GCN: %__pow_sign1 = tail call fast half @llvm.copysign.f16(half %__exp2, half %x)
define half @test_pow_fast_f16__y_13(half %x) {
+; GCN-POSTLINK-LABEL: define half @test_pow_fast_f16__y_13(
+; GCN-POSTLINK-SAME: half [[X:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-POSTLINK-NEXT: [[__FABS:%.*]] = tail call fast half @llvm.fabs.f16(half [[X]])
+; GCN-POSTLINK-NEXT: [[__LOG2:%.*]] = tail call fast half @llvm.log2.f16(half [[__FABS]])
+; GCN-POSTLINK-NEXT: [[__YLOGX:%.*]] = fmul fast half [[__LOG2]], 1.300000e+01
+; GCN-POSTLINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) half @llvm.exp2.f16(half [[__YLOGX]])
+; GCN-POSTLINK-NEXT: [[__POW_SIGN1:%.*]] = tail call fast half @llvm.copysign.f16(half [[__EXP2]], half [[X]])
+; GCN-POSTLINK-NEXT: ret half [[__POW_SIGN1]]
+;
+; GCN-PRELINK-LABEL: define half @test_pow_fast_f16__y_13(
+; GCN-PRELINK-SAME: half [[X:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-PRELINK-NEXT: [[__FABS:%.*]] = tail call fast half @llvm.fabs.f16(half [[X]])
+; GCN-PRELINK-NEXT: [[__LOG2:%.*]] = tail call fast half @llvm.log2.f16(half [[__FABS]])
+; GCN-PRELINK-NEXT: [[__YLOGX:%.*]] = fmul fast half [[__LOG2]], 1.300000e+01
+; GCN-PRELINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) half @llvm.exp2.f16(half [[__YLOGX]])
+; GCN-PRELINK-NEXT: [[__POW_SIGN1:%.*]] = tail call fast half @llvm.copysign.f16(half [[__EXP2]], half [[X]])
+; GCN-PRELINK-NEXT: ret half [[__POW_SIGN1]]
+;
+; GCN-NATIVE-LABEL: define half @test_pow_fast_f16__y_13(
+; GCN-NATIVE-SAME: half [[X:%.*]]) local_unnamed_addr #[[ATTR3]] {
+; GCN-NATIVE-NEXT: [[__FABS:%.*]] = tail call fast half @llvm.fabs.f16(half [[X]])
+; GCN-NATIVE-NEXT: [[__LOG2:%.*]] = tail call fast half @llvm.log2.f16(half [[__FABS]])
+; GCN-NATIVE-NEXT: [[__YLOGX:%.*]] = fmul fast half [[__LOG2]], 1.300000e+01
+; GCN-NATIVE-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) half @llvm.exp2.f16(half [[__YLOGX]])
+; GCN-NATIVE-NEXT: [[__POW_SIGN1:%.*]] = tail call fast half @llvm.copysign.f16(half [[__EXP2]], half [[X]])
+; GCN-NATIVE-NEXT: ret half [[__POW_SIGN1]]
+;
%powr = tail call fast half @_Z3powDhDh(half %x, half 13.0)
ret half %powr
}
-; GCN-LABEL: define <2 x half> @test_pow_fast_v2f16__y_13(<2 x half> %x)
-; GCN: %__fabs = tail call fast <2 x half> @llvm.fabs.v2f16(<2 x half> %x)
-; GCN: %__log2 = tail call fast <2 x half> @llvm.log2.v2f16(<2 x half> %__fabs)
-; GCN: %__ylogx = fmul fast <2 x half> %__log2, splat (half 1.300000e+01)
-; GCN: %__exp2 = tail call fast nofpclass(nan ninf nzero nsub nnorm) <2 x half> @llvm.exp2.v2f16(<2 x half> %__ylogx)
-; GCN: %__pow_sign1 = tail call fast <2 x half> @llvm.copysign.v2f16(<2 x half> %__exp2, <2 x half> %x)
define <2 x half> @test_pow_fast_v2f16__y_13(<2 x half> %x) {
+; GCN-POSTLINK-LABEL: define <2 x half> @test_pow_fast_v2f16__y_13(
+; GCN-POSTLINK-SAME: <2 x half> [[X:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-POSTLINK-NEXT: [[__FABS:%.*]] = tail call fast <2 x half> @llvm.fabs.v2f16(<2 x half> [[X]])
+; GCN-POSTLINK-NEXT: [[__LOG2:%.*]] = tail call fast <2 x half> @llvm.log2.v2f16(<2 x half> [[__FABS]])
+; GCN-POSTLINK-NEXT: [[__YLOGX:%.*]] = fmul fast <2 x half> [[__LOG2]], splat (half 1.300000e+01)
+; GCN-POSTLINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) <2 x half> @llvm.exp2.v2f16(<2 x half> [[__YLOGX]])
+; GCN-POSTLINK-NEXT: [[__POW_SIGN1:%.*]] = tail call fast <2 x half> @llvm.copysign.v2f16(<2 x half> [[__EXP2]], <2 x half> [[X]])
+; GCN-POSTLINK-NEXT: ret <2 x half> [[__POW_SIGN1]]
+;
+; GCN-PRELINK-LABEL: define <2 x half> @test_pow_fast_v2f16__y_13(
+; GCN-PRELINK-SAME: <2 x half> [[X:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-PRELINK-NEXT: [[__FABS:%.*]] = tail call fast <2 x half> @llvm.fabs.v2f16(<2 x half> [[X]])
+; GCN-PRELINK-NEXT: [[__LOG2:%.*]] = tail call fast <2 x half> @llvm.log2.v2f16(<2 x half> [[__FABS]])
+; GCN-PRELINK-NEXT: [[__YLOGX:%.*]] = fmul fast <2 x half> [[__LOG2]], splat (half 1.300000e+01)
+; GCN-PRELINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) <2 x half> @llvm.exp2.v2f16(<2 x half> [[__YLOGX]])
+; GCN-PRELINK-NEXT: [[__POW_SIGN1:%.*]] = tail call fast <2 x half> @llvm.copysign.v2f16(<2 x half> [[__EXP2]], <2 x half> [[X]])
+; GCN-PRELINK-NEXT: ret <2 x half> [[__POW_SIGN1]]
+;
+; GCN-NATIVE-LABEL: define <2 x half> @test_pow_fast_v2f16__y_13(
+; GCN-NATIVE-SAME: <2 x half> [[X:%.*]]) local_unnamed_addr #[[ATTR3]] {
+; GCN-NATIVE-NEXT: [[__FABS:%.*]] = tail call fast <2 x half> @llvm.fabs.v2f16(<2 x half> [[X]])
+; GCN-NATIVE-NEXT: [[__LOG2:%.*]] = tail call fast <2 x half> @llvm.log2.v2f16(<2 x half> [[__FABS]])
+; GCN-NATIVE-NEXT: [[__YLOGX:%.*]] = fmul fast <2 x half> [[__LOG2]], splat (half 1.300000e+01)
+; GCN-NATIVE-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) <2 x half> @llvm.exp2.v2f16(<2 x half> [[__YLOGX]])
+; GCN-NATIVE-NEXT: [[__POW_SIGN1:%.*]] = tail call fast <2 x half> @llvm.copysign.v2f16(<2 x half> [[__EXP2]], <2 x half> [[X]])
+; GCN-NATIVE-NEXT: ret <2 x half> [[__POW_SIGN1]]
+;
%powr = tail call fast <2 x half> @_Z3powDv2_DhS_(<2 x half> %x, <2 x half> <half 13.0, half 13.0>)
ret <2 x half> %powr
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_rootn_1
-; GCN: %tmp = load float, ptr addrspace(1) %arrayidx, align 4
-; GCN: store float %tmp, ptr addrspace(1) %a, align 4
define amdgpu_kernel void @test_rootn_1(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_rootn_1(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-POSTLINK-NEXT: store float [[TMP]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_rootn_1(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-PRELINK-NEXT: store float [[TMP]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_rootn_1(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((0, 4)) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-NATIVE-NEXT: store float [[TMP]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%arrayidx = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
%tmp = load float, ptr addrspace(1) %arrayidx, align 4
@@ -469,9 +1225,31 @@ entry:
declare float @_Z5rootnfi(float, i32)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_rootn_2
-; GCN: call fast float @llvm.sqrt.f32(float %tmp)
define amdgpu_kernel void @test_rootn_2(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_rootn_2(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.sqrt.f32(float [[TMP]]), !fpmath [[META0:![0-9]+]]
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_rootn_2(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.sqrt.f32(float [[TMP]]), !fpmath [[META0:![0-9]+]]
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_rootn_2(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast float @llvm.sqrt.f32(float [[TMP]]), !fpmath [[META0:![0-9]+]]
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z5rootnfi(float %tmp, i32 2)
@@ -479,15 +1257,39 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_rootn_3
-; GCN-POSTLINK: call fast float @llvm.log2.f32
-; GCN-POSTLINK: fmul
-; GCN-POSTLINK: call fast float @llvm.exp2.f32
-; GCN-POSTLINK: select fast i1
-; GCN-POSTLINK: call fast float @llvm.copysign.f32
-
-; GCN-PRELINK: %__rootn2cbrt = tail call fast float @_Z4cbrtf(float %tmp)
define amdgpu_kernel void @test_rootn_3(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_rootn_3(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = tail call fast float @llvm.fabs.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = tail call fast float @llvm.log2.f32(float [[TMP0]])
+; GCN-POSTLINK-NEXT: [[TMP2:%.*]] = fmul fast float [[TMP1]], f0x3EAAAAAB
+; GCN-POSTLINK-NEXT: [[TMP3:%.*]] = tail call fast float @llvm.exp2.f32(float [[TMP2]])
+; GCN-POSTLINK-NEXT: [[TMP4:%.*]] = tail call fast float @llvm.copysign.f32(float [[TMP3]], float [[TMP]])
+; GCN-POSTLINK-NEXT: [[TMP5:%.*]] = fcmp fast oeq float [[TMP]], 0.000000e+00
+; GCN-POSTLINK-NEXT: [[TMP6:%.*]] = select fast i1 [[TMP5]], float 0.000000e+00, float +inf
+; GCN-POSTLINK-NEXT: [[TMP7:%.*]] = tail call fast float @llvm.copysign.f32(float [[TMP6]], float [[TMP]])
+; GCN-POSTLINK-NEXT: [[TMP8:%.*]] = select fast i1 [[TMP5]], float [[TMP7]], float [[TMP4]]
+; GCN-POSTLINK-NEXT: store float [[TMP8]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_rootn_3(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR3:[0-9]+]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[__ROOTN2CBRT:%.*]] = tail call fast float @_Z4cbrtf(float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[__ROOTN2CBRT]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_rootn_3(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[__ROOTN2CBRT:%.*]] = tail call fast float @_Z4cbrtf(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[__ROOTN2CBRT]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z5rootnfi(float %tmp, i32 3)
@@ -495,9 +1297,31 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_rootn_m1
-; GCN: fdiv fast float 1.000000e+00, %tmp
define amdgpu_kernel void @test_rootn_m1(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_rootn_m1(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[__ROOTN2DIV:%.*]] = fdiv fast float 1.000000e+00, [[TMP]]
+; GCN-POSTLINK-NEXT: store float [[__ROOTN2DIV]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_rootn_m1(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[__ROOTN2DIV:%.*]] = fdiv fast float 1.000000e+00, [[TMP]]
+; GCN-PRELINK-NEXT: store float [[__ROOTN2DIV]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_rootn_m1(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[__ROOTN2DIV:%.*]] = fdiv fast float 1.000000e+00, [[TMP]]
+; GCN-NATIVE-NEXT: store float [[__ROOTN2DIV]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z5rootnfi(float %tmp, i32 -1)
@@ -505,10 +1329,34 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_rootn_m2
-; GCN: [[SQRT:%.+]] = tail call fast float @llvm.sqrt.f32(float %tmp)
-; GCN-NEXT: fdiv fast float 1.000000e+00, [[SQRT]]
define amdgpu_kernel void @test_rootn_m2(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_rootn_m2(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = tail call fast float @llvm.sqrt.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = fdiv fast float 1.000000e+00, [[TMP0]], !fpmath [[META0]]
+; GCN-POSTLINK-NEXT: store float [[TMP1]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_rootn_m2(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = tail call fast float @llvm.sqrt.f32(float [[TMP]])
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = fdiv fast float 1.000000e+00, [[TMP0]], !fpmath [[META0]]
+; GCN-PRELINK-NEXT: store float [[TMP1]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_rootn_m2(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[TMP0:%.*]] = tail call fast float @llvm.sqrt.f32(float [[TMP]])
+; GCN-NATIVE-NEXT: [[TMP1:%.*]] = fdiv fast float 1.000000e+00, [[TMP0]], !fpmath [[META0]]
+; GCN-NATIVE-NEXT: store float [[TMP1]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z5rootnfi(float %tmp, i32 -2)
@@ -516,9 +1364,25 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_fma_0x
-; GCN: store float %y
define amdgpu_kernel void @test_fma_0x(ptr addrspace(1) nocapture %a, float %y) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_fma_0x(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: store float [[Y]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_fma_0x(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: store float [[Y]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_fma_0x(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: store float [[Y]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3fmafff(float 0.000000e+00, float %tmp, float %y)
@@ -528,9 +1392,25 @@ entry:
declare float @_Z3fmafff(float, float, float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_fma_x0
-; GCN: store float %y,
define amdgpu_kernel void @test_fma_x0(ptr addrspace(1) nocapture %a, float %y) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_fma_x0(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: store float [[Y]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_fma_x0(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: store float [[Y]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_fma_x0(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: store float [[Y]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3fmafff(float %tmp, float 0.000000e+00, float %y)
@@ -538,9 +1418,25 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_mad_0x
-; GCN: store float %y,
define amdgpu_kernel void @test_mad_0x(ptr addrspace(1) nocapture %a, float %y) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_mad_0x(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: store float [[Y]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_mad_0x(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: store float [[Y]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_mad_0x(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: store float [[Y]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3madfff(float 0.000000e+00, float %tmp, float %y)
@@ -550,9 +1446,25 @@ entry:
declare float @_Z3madfff(float, float, float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_mad_x0
-; GCN: store float %y,
define amdgpu_kernel void @test_mad_x0(ptr addrspace(1) nocapture %a, float %y) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_mad_x0(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: store float [[Y]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_mad_x0(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: store float [[Y]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_mad_x0(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree writeonly captures(none) initializes((0, 4)) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: store float [[Y]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3madfff(float %tmp, float 0.000000e+00, float %y)
@@ -560,9 +1472,31 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_fma_x1y
-; GCN: %call = fadd fast float %tmp, %y
define amdgpu_kernel void @test_fma_x1y(ptr addrspace(1) nocapture %a, float %y) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_fma_x1y(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = fadd fast float [[TMP]], [[Y]]
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_fma_x1y(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = fadd fast float [[TMP]], [[Y]]
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_fma_x1y(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = fadd fast float [[TMP]], [[Y]]
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3fmafff(float %tmp, float 1.000000e+00, float %y)
@@ -570,9 +1504,31 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_fma_1xy
-; GCN: %call = fadd fast float %tmp, %y
define amdgpu_kernel void @test_fma_1xy(ptr addrspace(1) nocapture %a, float %y) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_fma_1xy(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = fadd fast float [[TMP]], [[Y]]
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_fma_1xy(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = fadd fast float [[TMP]], [[Y]]
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_fma_1xy(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]], float [[Y:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = fadd fast float [[TMP]], [[Y]]
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3fmafff(float 1.000000e+00, float %tmp, float %y)
@@ -580,21 +1536,71 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_fma_xy0
-; GCN: %call = fmul fast float %tmp1, %tmp
define amdgpu_kernel void @test_fma_xy0(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_fma_xy0(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = fmul fast float [[TMP1]], [[TMP]]
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_fma_xy0(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = fmul fast float [[TMP1]], [[TMP]]
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_fma_xy0(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX]], align 4
+; GCN-NATIVE-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = fmul fast float [[TMP1]], [[TMP]]
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%arrayidx = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
%tmp = load float, ptr addrspace(1) %arrayidx, align 4
- %tmp1 = load float, ptr addrspace(1) %a, align 4
- %call = call fast float @_Z3fmafff(float %tmp, float %tmp1, float 0.000000e+00)
+ %val1 = load float, ptr addrspace(1) %a, align 4
+ %call = call fast float @_Z3fmafff(float %tmp, float %val1, float 0.000000e+00)
store float %call, ptr addrspace(1) %a, align 4
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_use_native_exp
-; GCN-NATIVE: call fast float @llvm.exp.f32(float %tmp)
define amdgpu_kernel void @test_use_native_exp(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_use_native_exp(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.exp.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_use_native_exp(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.exp.f32(float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_use_native_exp(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast float @llvm.exp.f32(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3expf(float %tmp)
@@ -604,9 +1610,31 @@ entry:
declare float @_Z3expf(float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_use_native_exp2
-; GCN-NATIVE: call fast float @llvm.exp2.f32(float %tmp)
define amdgpu_kernel void @test_use_native_exp2(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_use_native_exp2(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.exp2.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_use_native_exp2(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.exp2.f32(float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_use_native_exp2(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast float @llvm.exp2.f32(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z4exp2f(float %tmp)
@@ -616,9 +1644,31 @@ entry:
declare float @_Z4exp2f(float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_use_native_exp10
-; GCN-NATIVE: call fast float @_Z12native_exp10f(float %tmp)
define amdgpu_kernel void @test_use_native_exp10(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_use_native_exp10(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @_Z5exp10f(float [[TMP]])
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_use_native_exp10(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast float @_Z5exp10f(float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_use_native_exp10(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast float @_Z12native_exp10f(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z5exp10f(float %tmp)
@@ -628,9 +1678,31 @@ entry:
declare float @_Z5exp10f(float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_use_native_log
-; GCN-NATIVE: call fast float @llvm.log.f32(float %tmp)
define amdgpu_kernel void @test_use_native_log(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_use_native_log(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.log.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_use_native_log(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.log.f32(float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_use_native_log(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast float @llvm.log.f32(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3logf(float %tmp)
@@ -640,9 +1712,31 @@ entry:
declare float @_Z3logf(float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_use_native_log2
-; GCN-NATIVE: call fast float @llvm.log2.f32(float %tmp)
define amdgpu_kernel void @test_use_native_log2(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_use_native_log2(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.log2.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_use_native_log2(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.log2.f32(float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_use_native_log2(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast float @llvm.log2.f32(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z4log2f(float %tmp)
@@ -652,9 +1746,31 @@ entry:
declare float @_Z4log2f(float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_use_native_log10
-; GCN-NATIVE: call fast float @llvm.log10.f32(float %tmp)
define amdgpu_kernel void @test_use_native_log10(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_use_native_log10(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.log10.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_use_native_log10(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.log10.f32(float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_use_native_log10(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast float @llvm.log10.f32(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z5log10f(float %tmp)
@@ -664,37 +1780,117 @@ entry:
declare float @_Z5log10f(float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_use_native_powr
-; GCN: %tmp1 = load float, ptr addrspace(1) %arrayidx1, align 4
-; GCN: %__log2 = tail call fast float @llvm.log2.f32(float %tmp)
-; GCN: %__ylogx = fmul fast float %tmp1, %__log2
-; GCN: %__exp2 = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float %__ylogx)
-; GCN: store float %__exp2, ptr addrspace(1) %a, align 4
define amdgpu_kernel void @test_use_native_powr(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_use_native_powr(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-POSTLINK-NEXT: [[__LOG2:%.*]] = tail call fast float @llvm.log2.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: [[__YLOGX:%.*]] = fmul fast float [[TMP1]], [[__LOG2]]
+; GCN-POSTLINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float [[__YLOGX]])
+; GCN-POSTLINK-NEXT: store float [[__EXP2]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_use_native_powr(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-PRELINK-NEXT: [[__LOG2:%.*]] = tail call fast float @llvm.log2.f32(float [[TMP]])
+; GCN-PRELINK-NEXT: [[__YLOGX:%.*]] = fmul fast float [[TMP1]], [[__LOG2]]
+; GCN-PRELINK-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float [[__YLOGX]])
+; GCN-PRELINK-NEXT: store float [[__EXP2]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_use_native_powr(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-NATIVE-NEXT: [[__LOG2:%.*]] = tail call fast float @llvm.log2.f32(float [[TMP]])
+; GCN-NATIVE-NEXT: [[__YLOGX:%.*]] = fmul fast float [[TMP1]], [[__LOG2]]
+; GCN-NATIVE-NEXT: [[__EXP2:%.*]] = tail call fast nofpclass(nan ninf nzero nsub nnorm) float @llvm.exp2.f32(float [[__YLOGX]])
+; GCN-NATIVE-NEXT: store float [[__EXP2]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%arrayidx1 = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
- %tmp1 = load float, ptr addrspace(1) %arrayidx1, align 4
- %call = call fast float @_Z4powrff(float %tmp, float %tmp1)
+ %val1 = load float, ptr addrspace(1) %arrayidx1, align 4
+ %call = call fast float @_Z4powrff(float %tmp, float %val1)
store float %call, ptr addrspace(1) %a, align 4
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_use_native_powr_nobuiltin
-; GCN: %call = tail call fast float @_Z4powrff(float %tmp, float %tmp1)
define amdgpu_kernel void @test_use_native_powr_nobuiltin(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_use_native_powr_nobuiltin(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @_Z4powrff(float [[TMP]], float [[TMP1]]) #[[ATTR5:[0-9]+]]
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_use_native_powr_nobuiltin(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast float @_Z4powrff(float [[TMP]], float [[TMP1]]) #[[ATTR7:[0-9]+]]
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_use_native_powr_nobuiltin(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[TMP1:%.*]] = load float, ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast float @_Z4powrff(float [[TMP]], float [[TMP1]]) #[[ATTR7:[0-9]+]]
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%arrayidx1 = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
- %tmp1 = load float, ptr addrspace(1) %arrayidx1, align 4
- %call = call fast float @_Z4powrff(float %tmp, float %tmp1) nobuiltin
+ %val1 = load float, ptr addrspace(1) %arrayidx1, align 4
+ %call = call fast float @_Z4powrff(float %tmp, float %val1) nobuiltin
store float %call, ptr addrspace(1) %a, align 4
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_use_native_sqrt
-; GCN-NATIVE: call fast float @llvm.sqrt.f32(float %tmp)
define amdgpu_kernel void @test_use_native_sqrt(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_use_native_sqrt(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.sqrt.f32(float [[TMP]])
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_use_native_sqrt(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast float @llvm.sqrt.f32(float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_use_native_sqrt(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast float @llvm.sqrt.f32(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z4sqrtf(float %tmp)
@@ -702,9 +1898,31 @@ entry:
ret void
}
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_dont_use_native_sqrt_fast_f64
-; GCN: call fast double @llvm.sqrt.f64(double %tmp)
define amdgpu_kernel void @test_dont_use_native_sqrt_fast_f64(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_dont_use_native_sqrt_fast_f64(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load double, ptr addrspace(1) [[A]], align 8
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast double @llvm.sqrt.f64(double [[TMP]])
+; GCN-POSTLINK-NEXT: store double [[CALL]], ptr addrspace(1) [[A]], align 8
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_dont_use_native_sqrt_fast_f64(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR1]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load double, ptr addrspace(1) [[A]], align 8
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast double @llvm.sqrt.f64(double [[TMP]])
+; GCN-PRELINK-NEXT: store double [[CALL]], ptr addrspace(1) [[A]], align 8
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_dont_use_native_sqrt_fast_f64(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR2]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load double, ptr addrspace(1) [[A]], align 8
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast double @llvm.sqrt.f64(double [[TMP]])
+; GCN-NATIVE-NEXT: store double [[CALL]], ptr addrspace(1) [[A]], align 8
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load double, ptr addrspace(1) %a, align 8
%call = call fast double @_Z4sqrtd(double %tmp)
@@ -715,9 +1933,31 @@ entry:
declare float @_Z4sqrtf(float)
declare double @_Z4sqrtd(double)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_use_native_rsqrt
-; GCN-NATIVE: call fast float @_Z12native_rsqrtf(float %tmp)
define amdgpu_kernel void @test_use_native_rsqrt(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_use_native_rsqrt(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @_Z5rsqrtf(float [[TMP]])
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_use_native_rsqrt(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast float @_Z5rsqrtf(float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_use_native_rsqrt(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast float @_Z12native_rsqrtf(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z5rsqrtf(float %tmp)
@@ -727,9 +1967,31 @@ entry:
declare float @_Z5rsqrtf(float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_use_native_tan
-; GCN-NATIVE: call fast float @_Z10native_tanf(float %tmp)
define amdgpu_kernel void @test_use_native_tan(ptr addrspace(1) nocapture %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_use_native_tan(
+; GCN-POSTLINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @_Z3tanf(float [[TMP]])
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_use_native_tan(
+; GCN-PRELINK-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast float @_Z3tanf(float [[TMP]])
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_use_native_tan(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[CALL:%.*]] = tail call fast float @_Z10native_tanf(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%call = call fast float @_Z3tanf(float %tmp)
@@ -739,15 +2001,43 @@ entry:
declare float @_Z3tanf(float)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_use_native_sincos
-; GCN-NATIVE: call float @_Z10native_sinf(float %tmp)
-; GCN-NATIVE: call float @_Z10native_cosf(float %tmp)
define amdgpu_kernel void @test_use_native_sincos(ptr addrspace(1) %a) {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_use_native_sincos(
+; GCN-POSTLINK-SAME: ptr addrspace(1) [[A:%.*]]) local_unnamed_addr {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = addrspacecast ptr addrspace(1) [[ARRAYIDX1]] to ptr
+; GCN-POSTLINK-NEXT: [[CALL:%.*]] = tail call fast float @_Z6sincosfPf(float [[TMP]], ptr [[TMP1]])
+; GCN-POSTLINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_use_native_sincos(
+; GCN-PRELINK-SAME: ptr addrspace(1) [[A:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = addrspacecast ptr addrspace(1) [[ARRAYIDX1]] to ptr
+; GCN-PRELINK-NEXT: [[CALL:%.*]] = tail call fast float @_Z6sincosfPf(float [[TMP]], ptr [[TMP1]])
+; GCN-PRELINK-NEXT: store float [[CALL]], ptr addrspace(1) [[A]], align 4
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_use_native_sincos(
+; GCN-NATIVE-SAME: ptr addrspace(1) nofree captures(none) initializes((4, 8)) [[A:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = load float, ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr addrspace(1) [[A]], i64 4
+; GCN-NATIVE-NEXT: [[SPLITSIN:%.*]] = tail call float @_Z10native_sinf(float [[TMP]])
+; GCN-NATIVE-NEXT: [[SPLITCOS:%.*]] = tail call float @_Z10native_cosf(float [[TMP]])
+; GCN-NATIVE-NEXT: store float [[SPLITCOS]], ptr addrspace(1) [[ARRAYIDX1]], align 4
+; GCN-NATIVE-NEXT: store float [[SPLITSIN]], ptr addrspace(1) [[A]], align 4
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = load float, ptr addrspace(1) %a, align 4
%arrayidx1 = getelementptr inbounds float, ptr addrspace(1) %a, i64 1
- %tmp1 = addrspacecast ptr addrspace(1) %arrayidx1 to ptr
- %call = call fast float @_Z6sincosfPf(float %tmp, ptr %tmp1)
+ %val1 = addrspacecast ptr addrspace(1) %arrayidx1 to ptr
+ %call = call fast float @_Z6sincosfPf(float %tmp, ptr %val1)
store float %call, ptr addrspace(1) %a, align 4
ret void
}
@@ -757,16 +2047,43 @@ declare float @_Z6sincosfPf(float, ptr)
%opencl.pipe_t = type opaque
%opencl.reserve_id_t = type opaque
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_read_pipe(ptr addrspace(1) %p, ptr addrspace(1) %ptr)
-; GCN-PRELINK: call i32 @__read_pipe_2_4(ptr addrspace(1) %{{.*}}, ptr %{{.*}}) #[[$NOUNWIND:[0-9]+]]
-; GCN-PRELINK: call i32 @__read_pipe_4_4(ptr addrspace(1) %{{.*}}, ptr addrspace(5) %{{.*}}, i32 2, ptr %{{.*}}) #[[$NOUNWIND]]
define amdgpu_kernel void @test_read_pipe(ptr addrspace(1) %p, ptr addrspace(1) %ptr) local_unnamed_addr {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_read_pipe(
+; GCN-POSTLINK-SAME: ptr addrspace(1) [[P:%.*]], ptr addrspace(1) [[PTR:%.*]]) local_unnamed_addr {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = addrspacecast ptr addrspace(1) [[PTR]] to ptr
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = tail call i32 @__read_pipe_2_4(ptr addrspace(1) [[P]], ptr [[TMP1]]) #[[ATTR3:[0-9]+]]
+; GCN-POSTLINK-NEXT: [[TMP3:%.*]] = tail call ptr addrspace(5) @__reserve_read_pipe(ptr addrspace(1) [[P]], i32 2, i32 4, i32 4)
+; GCN-POSTLINK-NEXT: [[TMP2:%.*]] = tail call i32 @__read_pipe_4_4(ptr addrspace(1) [[P]], ptr addrspace(5) [[TMP3]], i32 2, ptr [[TMP1]]) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: tail call void @__commit_read_pipe(ptr addrspace(1) [[P]], ptr addrspace(5) [[TMP3]], i32 4, i32 4)
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_read_pipe(
+; GCN-PRELINK-SAME: ptr addrspace(1) [[P:%.*]], ptr addrspace(1) [[PTR:%.*]]) local_unnamed_addr {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = addrspacecast ptr addrspace(1) [[PTR]] to ptr
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = tail call i32 @__read_pipe_2_4(ptr addrspace(1) [[P]], ptr [[TMP1]]) #[[ATTR4:[0-9]+]]
+; GCN-PRELINK-NEXT: [[TMP3:%.*]] = tail call ptr addrspace(5) @__reserve_read_pipe(ptr addrspace(1) [[P]], i32 2, i32 4, i32 4)
+; GCN-PRELINK-NEXT: [[TMP2:%.*]] = tail call i32 @__read_pipe_4_4(ptr addrspace(1) [[P]], ptr addrspace(5) [[TMP3]], i32 2, ptr [[TMP1]]) #[[ATTR4]]
+; GCN-PRELINK-NEXT: tail call void @__commit_read_pipe(ptr addrspace(1) [[P]], ptr addrspace(5) [[TMP3]], i32 4, i32 4)
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_read_pipe(
+; GCN-NATIVE-SAME: ptr addrspace(1) [[P:%.*]], ptr addrspace(1) [[PTR:%.*]]) local_unnamed_addr {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP1:%.*]] = addrspacecast ptr addrspace(1) [[PTR]] to ptr
+; GCN-NATIVE-NEXT: [[TMP0:%.*]] = tail call i32 @__read_pipe_2_4(ptr addrspace(1) [[P]], ptr [[TMP1]]) #[[ATTR4:[0-9]+]]
+; GCN-NATIVE-NEXT: [[TMP3:%.*]] = tail call ptr addrspace(5) @__reserve_read_pipe(ptr addrspace(1) [[P]], i32 2, i32 4, i32 4)
+; GCN-NATIVE-NEXT: [[TMP2:%.*]] = tail call i32 @__read_pipe_4_4(ptr addrspace(1) [[P]], ptr addrspace(5) [[TMP3]], i32 2, ptr [[TMP1]]) #[[ATTR4]]
+; GCN-NATIVE-NEXT: tail call void @__commit_read_pipe(ptr addrspace(1) [[P]], ptr addrspace(5) [[TMP3]], i32 4, i32 4)
+; GCN-NATIVE-NEXT: ret void
+;
entry:
- %tmp1 = addrspacecast ptr addrspace(1) %ptr to ptr
- %tmp2 = call i32 @__read_pipe_2(ptr addrspace(1) %p, ptr %tmp1, i32 4, i32 4) #0
- %tmp3 = call ptr addrspace(5) @__reserve_read_pipe(ptr addrspace(1) %p, i32 2, i32 4, i32 4)
- %tmp4 = call i32 @__read_pipe_4(ptr addrspace(1) %p, ptr addrspace(5) %tmp3, i32 2, ptr %tmp1, i32 4, i32 4) #0
- call void @__commit_read_pipe(ptr addrspace(1) %p, ptr addrspace(5) %tmp3, i32 4, i32 4)
+ %val1 = addrspacecast ptr addrspace(1) %ptr to ptr
+ %val2 = call i32 @__read_pipe_2(ptr addrspace(1) %p, ptr %val1, i32 4, i32 4) #0
+ %val3 = call ptr addrspace(5) @__reserve_read_pipe(ptr addrspace(1) %p, i32 2, i32 4, i32 4)
+ %val4 = call i32 @__read_pipe_4(ptr addrspace(1) %p, ptr addrspace(5) %val3, i32 2, ptr %val1, i32 4, i32 4) #0
+ call void @__commit_read_pipe(ptr addrspace(1) %p, ptr addrspace(5) %val3, i32 4, i32 4)
ret void
}
@@ -778,16 +2095,43 @@ declare i32 @__read_pipe_4(ptr addrspace(1), ptr addrspace(5), i32, ptr, i32, i3
declare void @__commit_read_pipe(ptr addrspace(1), ptr addrspace(5), i32, i32)
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_write_pipe(ptr addrspace(1) %p, ptr addrspace(1) %ptr)
-; GCN-PRELINK: call i32 @__write_pipe_2_4(ptr addrspace(1) %{{.*}}, ptr %{{.*}}) #[[$NOUNWIND]]
-; GCN-PRELINK: call i32 @__write_pipe_4_4(ptr addrspace(1) %{{.*}}, ptr addrspace(5) %{{.*}}, i32 2, ptr %{{.*}}) #[[$NOUNWIND]]
define amdgpu_kernel void @test_write_pipe(ptr addrspace(1) %p, ptr addrspace(1) %ptr) local_unnamed_addr {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_write_pipe(
+; GCN-POSTLINK-SAME: ptr addrspace(1) [[P:%.*]], ptr addrspace(1) [[PTR:%.*]]) local_unnamed_addr #[[ATTR3]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = addrspacecast ptr addrspace(1) [[PTR]] to ptr
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = tail call i32 @__write_pipe_2_4(ptr addrspace(1) [[P]], ptr [[TMP1]]) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: [[VAL3:%.*]] = tail call ptr addrspace(5) @__reserve_write_pipe(ptr addrspace(1) [[P]], i32 2, i32 4, i32 4) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: [[TMP2:%.*]] = tail call i32 @__write_pipe_4_4(ptr addrspace(1) [[P]], ptr addrspace(5) [[VAL3]], i32 2, ptr [[TMP1]]) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: tail call void @__commit_write_pipe(ptr addrspace(1) [[P]], ptr addrspace(5) [[VAL3]], i32 4, i32 4) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_write_pipe(
+; GCN-PRELINK-SAME: ptr addrspace(1) [[P:%.*]], ptr addrspace(1) [[PTR:%.*]]) local_unnamed_addr #[[ATTR4]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = addrspacecast ptr addrspace(1) [[PTR]] to ptr
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = tail call i32 @__write_pipe_2_4(ptr addrspace(1) [[P]], ptr [[TMP1]]) #[[ATTR4]]
+; GCN-PRELINK-NEXT: [[VAL3:%.*]] = tail call ptr addrspace(5) @__reserve_write_pipe(ptr addrspace(1) [[P]], i32 2, i32 4, i32 4) #[[ATTR4]]
+; GCN-PRELINK-NEXT: [[TMP2:%.*]] = tail call i32 @__write_pipe_4_4(ptr addrspace(1) [[P]], ptr addrspace(5) [[VAL3]], i32 2, ptr [[TMP1]]) #[[ATTR4]]
+; GCN-PRELINK-NEXT: tail call void @__commit_write_pipe(ptr addrspace(1) [[P]], ptr addrspace(5) [[VAL3]], i32 4, i32 4) #[[ATTR4]]
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_write_pipe(
+; GCN-NATIVE-SAME: ptr addrspace(1) [[P:%.*]], ptr addrspace(1) [[PTR:%.*]]) local_unnamed_addr #[[ATTR4]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP1:%.*]] = addrspacecast ptr addrspace(1) [[PTR]] to ptr
+; GCN-NATIVE-NEXT: [[TMP0:%.*]] = tail call i32 @__write_pipe_2_4(ptr addrspace(1) [[P]], ptr [[TMP1]]) #[[ATTR4]]
+; GCN-NATIVE-NEXT: [[VAL3:%.*]] = tail call ptr addrspace(5) @__reserve_write_pipe(ptr addrspace(1) [[P]], i32 2, i32 4, i32 4) #[[ATTR4]]
+; GCN-NATIVE-NEXT: [[TMP2:%.*]] = tail call i32 @__write_pipe_4_4(ptr addrspace(1) [[P]], ptr addrspace(5) [[VAL3]], i32 2, ptr [[TMP1]]) #[[ATTR4]]
+; GCN-NATIVE-NEXT: tail call void @__commit_write_pipe(ptr addrspace(1) [[P]], ptr addrspace(5) [[VAL3]], i32 4, i32 4) #[[ATTR4]]
+; GCN-NATIVE-NEXT: ret void
+;
entry:
- %tmp1 = addrspacecast ptr addrspace(1) %ptr to ptr
- %tmp2 = call i32 @__write_pipe_2(ptr addrspace(1) %p, ptr %tmp1, i32 4, i32 4) #0
- %tmp3 = call ptr addrspace(5) @__reserve_write_pipe(ptr addrspace(1) %p, i32 2, i32 4, i32 4) #0
- %tmp4 = call i32 @__write_pipe_4(ptr addrspace(1) %p, ptr addrspace(5) %tmp3, i32 2, ptr %tmp1, i32 4, i32 4) #0
- call void @__commit_write_pipe(ptr addrspace(1) %p, ptr addrspace(5) %tmp3, i32 4, i32 4) #0
+ %val1 = addrspacecast ptr addrspace(1) %ptr to ptr
+ %val2 = call i32 @__write_pipe_2(ptr addrspace(1) %p, ptr %val1, i32 4, i32 4) #0
+ %val3 = call ptr addrspace(5) @__reserve_write_pipe(ptr addrspace(1) %p, i32 2, i32 4, i32 4) #0
+ %val4 = call i32 @__write_pipe_4(ptr addrspace(1) %p, ptr addrspace(5) %val3, i32 2, ptr %val1, i32 4, i32 4) #0
+ call void @__commit_write_pipe(ptr addrspace(1) %p, ptr addrspace(5) %val3, i32 4, i32 4) #0
ret void
}
@@ -801,41 +2145,103 @@ declare void @__commit_write_pipe(ptr addrspace(1), ptr addrspace(5), i32, i32)
%struct.S = type { [100 x i32] }
-; GCN-LABEL: {{^}}define amdgpu_kernel void @test_pipe_size
-; GCN-PRELINK: call i32 @__read_pipe_2_1(ptr addrspace(1) %{{.*}} ptr %{{.*}}) #[[$NOUNWIND]]
-; GCN-PRELINK: call i32 @__read_pipe_2_2(ptr addrspace(1) %{{.*}} ptr %{{.*}}) #[[$NOUNWIND]]
-; GCN-PRELINK: call i32 @__read_pipe_2_4(ptr addrspace(1) %{{.*}} ptr %{{.*}}) #[[$NOUNWIND]]
-; GCN-PRELINK: call i32 @__read_pipe_2_8(ptr addrspace(1) %{{.*}} ptr %{{.*}}) #[[$NOUNWIND]]
-; GCN-PRELINK: call i32 @__read_pipe_2_16(ptr addrspace(1) %{{.*}}, ptr %{{.*}}) #[[$NOUNWIND]]
-; GCN-PRELINK: call i32 @__read_pipe_2_32(ptr addrspace(1) %{{.*}}, ptr %{{.*}} #[[$NOUNWIND]]
-; GCN-PRELINK: call i32 @__read_pipe_2_64(ptr addrspace(1) %{{.*}}, ptr %{{.*}} #[[$NOUNWIND]]
-; GCN-PRELINK: call i32 @__read_pipe_2_128(ptr addrspace(1) %{{.*}}, ptr %{{.*}} #[[$NOUNWIND]]
-; GCN-PRELINK: call i32 @__read_pipe_2(ptr addrspace(1) %{{.*}}, ptr %{{.*}} i32 400, i32 4) #[[$NOUNWIND]]
define amdgpu_kernel void @test_pipe_size(ptr addrspace(1) %p1, ptr addrspace(1) %ptr1, ptr addrspace(1) %p2, ptr addrspace(1) %ptr2, ptr addrspace(1) %p4, ptr addrspace(1) %ptr4, ptr addrspace(1) %p8, ptr addrspace(1) %ptr8, ptr addrspace(1) %p16, ptr addrspace(1) %ptr16, ptr addrspace(1) %p32, ptr addrspace(1) %ptr32, ptr addrspace(1) %p64, ptr addrspace(1) %ptr64, ptr addrspace(1) %p128, ptr addrspace(1) %ptr128, ptr addrspace(1) %pu, ptr addrspace(1) %ptru) local_unnamed_addr #0 {
+; GCN-POSTLINK-LABEL: define amdgpu_kernel void @test_pipe_size(
+; GCN-POSTLINK-SAME: ptr addrspace(1) [[P1:%.*]], ptr addrspace(1) [[PTR1:%.*]], ptr addrspace(1) [[P2:%.*]], ptr addrspace(1) [[PTR2:%.*]], ptr addrspace(1) [[P4:%.*]], ptr addrspace(1) [[PTR4:%.*]], ptr addrspace(1) [[P8:%.*]], ptr addrspace(1) [[PTR8:%.*]], ptr addrspace(1) [[P16:%.*]], ptr addrspace(1) [[PTR16:%.*]], ptr addrspace(1) [[P32:%.*]], ptr addrspace(1) [[PTR32:%.*]], ptr addrspace(1) [[P64:%.*]], ptr addrspace(1) [[PTR64:%.*]], ptr addrspace(1) [[P128:%.*]], ptr addrspace(1) [[PTR128:%.*]], ptr addrspace(1) [[PU:%.*]], ptr addrspace(1) [[PTRU:%.*]]) local_unnamed_addr #[[ATTR3]] {
+; GCN-POSTLINK-NEXT: [[ENTRY:.*:]]
+; GCN-POSTLINK-NEXT: [[TMP:%.*]] = addrspacecast ptr addrspace(1) [[PTR1]] to ptr
+; GCN-POSTLINK-NEXT: [[TMP0:%.*]] = tail call i32 @__read_pipe_2_1(ptr addrspace(1) [[P1]], ptr [[TMP]]) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: [[TMP3:%.*]] = addrspacecast ptr addrspace(1) [[PTR2]] to ptr
+; GCN-POSTLINK-NEXT: [[TMP1:%.*]] = tail call i32 @__read_pipe_2_2(ptr addrspace(1) [[P2]], ptr [[TMP3]]) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: [[TMP6:%.*]] = addrspacecast ptr addrspace(1) [[PTR4]] to ptr
+; GCN-POSTLINK-NEXT: [[TMP2:%.*]] = tail call i32 @__read_pipe_2_4(ptr addrspace(1) [[P4]], ptr [[TMP6]]) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: [[TMP9:%.*]] = addrspacecast ptr addrspace(1) [[PTR8]] to ptr
+; GCN-POSTLINK-NEXT: [[TMP8:%.*]] = tail call i32 @__read_pipe_2_8(ptr addrspace(1) [[P8]], ptr [[TMP9]]) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: [[TMP12:%.*]] = addrspacecast ptr addrspace(1) [[PTR16]] to ptr
+; GCN-POSTLINK-NEXT: [[TMP4:%.*]] = tail call i32 @__read_pipe_2_16(ptr addrspace(1) [[P16]], ptr [[TMP12]]) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: [[TMP15:%.*]] = addrspacecast ptr addrspace(1) [[PTR32]] to ptr
+; GCN-POSTLINK-NEXT: [[TMP5:%.*]] = tail call i32 @__read_pipe_2_32(ptr addrspace(1) [[P32]], ptr [[TMP15]]) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: [[TMP18:%.*]] = addrspacecast ptr addrspace(1) [[PTR64]] to ptr
+; GCN-POSTLINK-NEXT: [[TMP10:%.*]] = tail call i32 @__read_pipe_2_64(ptr addrspace(1) [[P64]], ptr [[TMP18]]) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: [[TMP21:%.*]] = addrspacecast ptr addrspace(1) [[PTR128]] to ptr
+; GCN-POSTLINK-NEXT: [[TMP7:%.*]] = tail call i32 @__read_pipe_2_128(ptr addrspace(1) [[P128]], ptr [[TMP21]]) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: [[TMP24:%.*]] = addrspacecast ptr addrspace(1) [[PTRU]] to ptr
+; GCN-POSTLINK-NEXT: [[VAL25:%.*]] = tail call i32 @__read_pipe_2(ptr addrspace(1) [[PU]], ptr [[TMP24]], i32 400, i32 4) #[[ATTR3]]
+; GCN-POSTLINK-NEXT: ret void
+;
+; GCN-PRELINK-LABEL: define amdgpu_kernel void @test_pipe_size(
+; GCN-PRELINK-SAME: ptr addrspace(1) [[P1:%.*]], ptr addrspace(1) [[PTR1:%.*]], ptr addrspace(1) [[P2:%.*]], ptr addrspace(1) [[PTR2:%.*]], ptr addrspace(1) [[P4:%.*]], ptr addrspace(1) [[PTR4:%.*]], ptr addrspace(1) [[P8:%.*]], ptr addrspace(1) [[PTR8:%.*]], ptr addrspace(1) [[P16:%.*]], ptr addrspace(1) [[PTR16:%.*]], ptr addrspace(1) [[P32:%.*]], ptr addrspace(1) [[PTR32:%.*]], ptr addrspace(1) [[P64:%.*]], ptr addrspace(1) [[PTR64:%.*]], ptr addrspace(1) [[P128:%.*]], ptr addrspace(1) [[PTR128:%.*]], ptr addrspace(1) [[PU:%.*]], ptr addrspace(1) [[PTRU:%.*]]) local_unnamed_addr #[[ATTR4]] {
+; GCN-PRELINK-NEXT: [[ENTRY:.*:]]
+; GCN-PRELINK-NEXT: [[TMP:%.*]] = addrspacecast ptr addrspace(1) [[PTR1]] to ptr
+; GCN-PRELINK-NEXT: [[TMP0:%.*]] = tail call i32 @__read_pipe_2_1(ptr addrspace(1) [[P1]], ptr [[TMP]]) #[[ATTR4]]
+; GCN-PRELINK-NEXT: [[TMP3:%.*]] = addrspacecast ptr addrspace(1) [[PTR2]] to ptr
+; GCN-PRELINK-NEXT: [[TMP1:%.*]] = tail call i32 @__read_pipe_2_2(ptr addrspace(1) [[P2]], ptr [[TMP3]]) #[[ATTR4]]
+; GCN-PRELINK-NEXT: [[TMP6:%.*]] = addrspacecast ptr addrspace(1) [[PTR4]] to ptr
+; GCN-PRELINK-NEXT: [[TMP2:%.*]] = tail call i32 @__read_pipe_2_4(ptr addrspace(1) [[P4]], ptr [[TMP6]]) #[[ATTR4]]
+; GCN-PRELINK-NEXT: [[TMP9:%.*]] = addrspacecast ptr addrspace(1) [[PTR8]] to ptr
+; GCN-PRELINK-NEXT: [[TMP8:%.*]] = tail call i32 @__read_pipe_2_8(ptr addrspace(1) [[P8]], ptr [[TMP9]]) #[[ATTR4]]
+; GCN-PRELINK-NEXT: [[TMP12:%.*]] = addrspacecast ptr addrspace(1) [[PTR16]] to ptr
+; GCN-PRELINK-NEXT: [[TMP4:%.*]] = tail call i32 @__read_pipe_2_16(ptr addrspace(1) [[P16]], ptr [[TMP12]]) #[[ATTR4]]
+; GCN-PRELINK-NEXT: [[TMP15:%.*]] = addrspacecast ptr addrspace(1) [[PTR32]] to ptr
+; GCN-PRELINK-NEXT: [[TMP5:%.*]] = tail call i32 @__read_pipe_2_32(ptr addrspace(1) [[P32]], ptr [[TMP15]]) #[[ATTR4]]
+; GCN-PRELINK-NEXT: [[TMP18:%.*]] = addrspacecast ptr addrspace(1) [[PTR64]] to ptr
+; GCN-PRELINK-NEXT: [[TMP10:%.*]] = tail call i32 @__read_pipe_2_64(ptr addrspace(1) [[P64]], ptr [[TMP18]]) #[[ATTR4]]
+; GCN-PRELINK-NEXT: [[TMP21:%.*]] = addrspacecast ptr addrspace(1) [[PTR128]] to ptr
+; GCN-PRELINK-NEXT: [[TMP7:%.*]] = tail call i32 @__read_pipe_2_128(ptr addrspace(1) [[P128]], ptr [[TMP21]]) #[[ATTR4]]
+; GCN-PRELINK-NEXT: [[TMP24:%.*]] = addrspacecast ptr addrspace(1) [[PTRU]] to ptr
+; GCN-PRELINK-NEXT: [[VAL25:%.*]] = tail call i32 @__read_pipe_2(ptr addrspace(1) [[PU]], ptr [[TMP24]], i32 400, i32 4) #[[ATTR4]]
+; GCN-PRELINK-NEXT: ret void
+;
+; GCN-NATIVE-LABEL: define amdgpu_kernel void @test_pipe_size(
+; GCN-NATIVE-SAME: ptr addrspace(1) [[P1:%.*]], ptr addrspace(1) [[PTR1:%.*]], ptr addrspace(1) [[P2:%.*]], ptr addrspace(1) [[PTR2:%.*]], ptr addrspace(1) [[P4:%.*]], ptr addrspace(1) [[PTR4:%.*]], ptr addrspace(1) [[P8:%.*]], ptr addrspace(1) [[PTR8:%.*]], ptr addrspace(1) [[P16:%.*]], ptr addrspace(1) [[PTR16:%.*]], ptr addrspace(1) [[P32:%.*]], ptr addrspace(1) [[PTR32:%.*]], ptr addrspace(1) [[P64:%.*]], ptr addrspace(1) [[PTR64:%.*]], ptr addrspace(1) [[P128:%.*]], ptr addrspace(1) [[PTR128:%.*]], ptr addrspace(1) [[PU:%.*]], ptr addrspace(1) [[PTRU:%.*]]) local_unnamed_addr #[[ATTR4]] {
+; GCN-NATIVE-NEXT: [[ENTRY:.*:]]
+; GCN-NATIVE-NEXT: [[TMP:%.*]] = addrspacecast ptr addrspace(1) [[PTR1]] to ptr
+; GCN-NATIVE-NEXT: [[TMP0:%.*]] = tail call i32 @__read_pipe_2_1(ptr addrspace(1) [[P1]], ptr [[TMP]]) #[[ATTR4]]
+; GCN-NATIVE-NEXT: [[TMP3:%.*]] = addrspacecast ptr addrspace(1) [[PTR2]] to ptr
+; GCN-NATIVE-NEXT: [[TMP1:%.*]] = tail call i32 @__read_pipe_2_2(ptr addrspace(1) [[P2]], ptr [[TMP3]]) #[[ATTR4]]
+; GCN-NATIVE-NEXT: [[TMP6:%.*]] = addrspacecast ptr addrspace(1) [[PTR4]] to ptr
+; GCN-NATIVE-NEXT: [[TMP2:%.*]] = tail call i32 @__read_pipe_2_4(ptr addrspace(1) [[P4]], ptr [[TMP6]]) #[[ATTR4]]
+; GCN-NATIVE-NEXT: [[TMP9:%.*]] = addrspacecast ptr addrspace(1) [[PTR8]] to ptr
+; GCN-NATIVE-NEXT: [[TMP8:%.*]] = tail call i32 @__read_pipe_2_8(ptr addrspace(1) [[P8]], ptr [[TMP9]]) #[[ATTR4]]
+; GCN-NATIVE-NEXT: [[TMP12:%.*]] = addrspacecast ptr addrspace(1) [[PTR16]] to ptr
+; GCN-NATIVE-NEXT: [[TMP4:%.*]] = tail call i32 @__read_pipe_2_16(ptr addrspace(1) [[P16]], ptr [[TMP12]]) #[[ATTR4]]
+; GCN-NATIVE-NEXT: [[TMP15:%.*]] = addrspacecast ptr addrspace(1) [[PTR32]] to ptr
+; GCN-NATIVE-NEXT: [[TMP5:%.*]] = tail call i32 @__read_pipe_2_32(ptr addrspace(1) [[P32]], ptr [[TMP15]]) #[[ATTR4]]
+; GCN-NATIVE-NEXT: [[TMP18:%.*]] = addrspacecast ptr addrspace(1) [[PTR64]] to ptr
+; GCN-NATIVE-NEXT: [[TMP10:%.*]] = tail call i32 @__read_pipe_2_64(ptr addrspace(1) [[P64]], ptr [[TMP18]]) #[[ATTR4]]
+; GCN-NATIVE-NEXT: [[TMP21:%.*]] = addrspacecast ptr addrspace(1) [[PTR128]] to ptr
+; GCN-NATIVE-NEXT: [[TMP7:%.*]] = tail call i32 @__read_pipe_2_128(ptr addrspace(1) [[P128]], ptr [[TMP21]]) #[[ATTR4]]
+; GCN-NATIVE-NEXT: [[TMP24:%.*]] = addrspacecast ptr addrspace(1) [[PTRU]] to ptr
+; GCN-NATIVE-NEXT: [[VAL25:%.*]] = tail call i32 @__read_pipe_2(ptr addrspace(1) [[PU]], ptr [[TMP24]], i32 400, i32 4) #[[ATTR4]]
+; GCN-NATIVE-NEXT: ret void
+;
entry:
%tmp = addrspacecast ptr addrspace(1) %ptr1 to ptr
- %tmp1 = call i32 @__read_pipe_2(ptr addrspace(1) %p1, ptr %tmp, i32 1, i32 1) #0
- %tmp3 = addrspacecast ptr addrspace(1) %ptr2 to ptr
- %tmp4 = call i32 @__read_pipe_2(ptr addrspace(1) %p2, ptr %tmp3, i32 2, i32 2) #0
- %tmp6 = addrspacecast ptr addrspace(1) %ptr4 to ptr
- %tmp7 = call i32 @__read_pipe_2(ptr addrspace(1) %p4, ptr %tmp6, i32 4, i32 4) #0
- %tmp9 = addrspacecast ptr addrspace(1) %ptr8 to ptr
- %tmp10 = call i32 @__read_pipe_2(ptr addrspace(1) %p8, ptr %tmp9, i32 8, i32 8) #0
- %tmp12 = addrspacecast ptr addrspace(1) %ptr16 to ptr
- %tmp13 = call i32 @__read_pipe_2(ptr addrspace(1) %p16, ptr %tmp12, i32 16, i32 16) #0
- %tmp15 = addrspacecast ptr addrspace(1) %ptr32 to ptr
- %tmp16 = call i32 @__read_pipe_2(ptr addrspace(1) %p32, ptr %tmp15, i32 32, i32 32) #0
- %tmp18 = addrspacecast ptr addrspace(1) %ptr64 to ptr
- %tmp19 = call i32 @__read_pipe_2(ptr addrspace(1) %p64, ptr %tmp18, i32 64, i32 64) #0
- %tmp21 = addrspacecast ptr addrspace(1) %ptr128 to ptr
- %tmp22 = call i32 @__read_pipe_2(ptr addrspace(1) %p128, ptr %tmp21, i32 128, i32 128) #0
- %tmp24 = addrspacecast ptr addrspace(1) %ptru to ptr
- %tmp25 = call i32 @__read_pipe_2(ptr addrspace(1) %pu, ptr %tmp24, i32 400, i32 4) #0
- ret void
-}
-
-; GCN-PRELINK: declare float @_Z4cbrtf(float) local_unnamed_addr #[[$NOUNWIND_READONLY:[0-9]+]]
-
-; GCN-PRELINK-DAG: attributes #[[$NOUNWIND]] = { nounwind }
-; GCN-PRELINK-DAG: attributes #[[$NOUNWIND_READONLY]] = { nounwind memory(read) }
+ %val1 = call i32 @__read_pipe_2(ptr addrspace(1) %p1, ptr %tmp, i32 1, i32 1) #0
+ %val3 = addrspacecast ptr addrspace(1) %ptr2 to ptr
+ %val4 = call i32 @__read_pipe_2(ptr addrspace(1) %p2, ptr %val3, i32 2, i32 2) #0
+ %val6 = addrspacecast ptr addrspace(1) %ptr4 to ptr
+ %val7 = call i32 @__read_pipe_2(ptr addrspace(1) %p4, ptr %val6, i32 4, i32 4) #0
+ %val9 = addrspacecast ptr addrspace(1) %ptr8 to ptr
+ %val10 = call i32 @__read_pipe_2(ptr addrspace(1) %p8, ptr %val9, i32 8, i32 8) #0
+ %val12 = addrspacecast ptr addrspace(1) %ptr16 to ptr
+ %val13 = call i32 @__read_pipe_2(ptr addrspace(1) %p16, ptr %val12, i32 16, i32 16) #0
+ %val15 = addrspacecast ptr addrspace(1) %ptr32 to ptr
+ %val16 = call i32 @__read_pipe_2(ptr addrspace(1) %p32, ptr %val15, i32 32, i32 32) #0
+ %val18 = addrspacecast ptr addrspace(1) %ptr64 to ptr
+ %val19 = call i32 @__read_pipe_2(ptr addrspace(1) %p64, ptr %val18, i32 64, i32 64) #0
+ %val21 = addrspacecast ptr addrspace(1) %ptr128 to ptr
+ %val22 = call i32 @__read_pipe_2(ptr addrspace(1) %p128, ptr %val21, i32 128, i32 128) #0
+ %val24 = addrspacecast ptr addrspace(1) %ptru to ptr
+ %val25 = call i32 @__read_pipe_2(ptr addrspace(1) %pu, ptr %val24, i32 400, i32 4) #0
+ ret void
+}
+
attributes #0 = { nounwind }
+;.
+; GCN-POSTLINK: [[META0]] = !{float 2.000000e+00}
+;.
+; GCN-PRELINK: [[META0]] = !{float 2.000000e+00}
+;.
+; GCN-NATIVE: [[META0]] = !{float 2.000000e+00}
+;.
More information about the llvm-branch-commits
mailing list