[llvm] [SLP][AMDGPU] Fix load-address typo in v4f16 combine tests (PR #222112)
Pablo Reble via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 08:47:46 PDT 2026
https://github.com/reble updated https://github.com/llvm/llvm-project/pull/222112
>From eee4e7578461911ed850db34e9b1016795680f38 Mon Sep 17 00:00:00 2001
From: Pablo Reble <pablo.reble at amd.com>
Date: Tue, 8 Sep 2026 12:03:57 -0500
Subject: [PATCH 1/3] [SLP][AMDGPU] Fix broken load-address typo in
slp-v2f16.ll v4f16 tests
---
.../SLPVectorizer/AMDGPU/slp-v2f16.ll | 81 +++++--------------
1 file changed, 22 insertions(+), 59 deletions(-)
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
index 498d9ffd7f594..a0b6b0e939ce1 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
@@ -465,30 +465,23 @@ bb:
; FIXME: Should always vectorize
define void @copysign_combine_v4f16(ptr addrspace(1) %arg, half %sign) {
-; GCN-LABEL: define void @copysign_combine_v4f16(
-; GCN-SAME: ptr addrspace(1) [[ARG:%.*]], half [[SIGN:%.*]]) {
-; GCN-NEXT: [[BB:.*:]]
-; GCN-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
-; GCN-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
-; GCN-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
-; GCN-NEXT: [[ITMP5:%.*]] = add nuw nsw i64 [[ITMP1]], 1
-; GCN-NEXT: [[ITMP6:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP5]]
-; GCN-NEXT: [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
-; GCN-NEXT: [[TMP1:%.*]] = insertelement <2 x half> poison, half [[SIGN]], i64 0
-; GCN-NEXT: [[TMP2:%.*]] = shufflevector <2 x half> [[TMP1]], <2 x half> poison, <2 x i32> zeroinitializer
-; GCN-NEXT: [[TMP3:%.*]] = call <2 x half> @llvm.copysign.v2f16(<2 x half> [[TMP0]], <2 x half> [[TMP2]])
-; GCN-NEXT: store <2 x half> [[TMP3]], ptr addrspace(1) [[ITMP2]], align 2
-; GCN-NEXT: [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
-; GCN-NEXT: [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
-; GCN-NEXT: [[ITMP11:%.*]] = load half, ptr addrspace(1) [[ITMP6]], align 2
-; GCN-NEXT: [[ITMP12:%.*]] = call half @llvm.copysign.f16(half [[ITMP11]], half [[SIGN]])
-; GCN-NEXT: store half [[ITMP12]], ptr addrspace(1) [[ITMP10]], align 2
-; GCN-NEXT: [[ITMP13:%.*]] = add nuw nsw i64 [[ITMP1]], 3
-; GCN-NEXT: [[ITMP14:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP13]]
-; GCN-NEXT: [[ITMP15:%.*]] = load half, ptr addrspace(1) [[ITMP14]], align 2
-; GCN-NEXT: [[ITMP16:%.*]] = call half @llvm.copysign.f16(half [[ITMP15]], half [[SIGN]])
-; GCN-NEXT: store half [[ITMP16]], ptr addrspace(1) [[ITMP14]], align 2
-; GCN-NEXT: ret void
+; GFX8-LABEL: define void @copysign_combine_v4f16(
+; GFX8-SAME: ptr addrspace(1) [[ARG:%.*]], half [[SIGN:%.*]]) {
+; GFX8-NEXT: [[BB:.*:]]
+; GFX8-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX8-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX8-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX8-NEXT: [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX8-NEXT: [[TMP1:%.*]] = insertelement <2 x half> poison, half [[SIGN]], i64 0
+; GFX8-NEXT: [[TMP2:%.*]] = shufflevector <2 x half> [[TMP1]], <2 x half> poison, <2 x i32> zeroinitializer
+; GFX8-NEXT: [[TMP3:%.*]] = call <2 x half> @llvm.copysign.v2f16(<2 x half> [[TMP0]], <2 x half> [[TMP2]])
+; GFX8-NEXT: store <2 x half> [[TMP3]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX8-NEXT: [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
+; GFX8-NEXT: [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
+; GFX8-NEXT: [[TMP4:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP10]], align 2
+; GFX8-NEXT: [[TMP5:%.*]] = call <2 x half> @llvm.copysign.v2f16(<2 x half> [[TMP4]], <2 x half> [[TMP2]])
+; GFX8-NEXT: store <2 x half> [[TMP5]], ptr addrspace(1) [[ITMP10]], align 2
+; GFX8-NEXT: ret void
;
bb:
%tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
@@ -507,7 +500,7 @@ bb:
%tmp9 = add nuw nsw i64 %tmp1, 2
%tmp10 = getelementptr inbounds half, ptr addrspace(1) %arg, i64 %tmp9
- %tmp11 = load half, ptr addrspace(1) %tmp6, align 2
+ %tmp11 = load half, ptr addrspace(1) %tmp10, align 2
%tmp12 = call half @llvm.copysign.f16(half %tmp11, half %sign)
store half %tmp12, ptr addrspace(1) %tmp10, align 2
@@ -527,46 +520,16 @@ define void @canonicalize_combine_v4f16(ptr addrspace(1) %arg) {
; GFX8-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
; GFX8-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
; GFX8-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
-; GFX8-NEXT: [[ITMP5:%.*]] = add nuw nsw i64 [[ITMP1]], 1
-; GFX8-NEXT: [[ITMP6:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP5]]
; GFX8-NEXT: [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
; GFX8-NEXT: [[TMP1:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP0]])
; GFX8-NEXT: store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
; GFX8-NEXT: [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
; GFX8-NEXT: [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
-; GFX8-NEXT: [[ITMP11:%.*]] = load half, ptr addrspace(1) [[ITMP6]], align 2
-; GFX8-NEXT: [[ITMP12:%.*]] = call half @llvm.canonicalize.f16(half [[ITMP11]])
-; GFX8-NEXT: store half [[ITMP12]], ptr addrspace(1) [[ITMP10]], align 2
-; GFX8-NEXT: [[ITMP13:%.*]] = add nuw nsw i64 [[ITMP1]], 3
-; GFX8-NEXT: [[ITMP14:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP13]]
-; GFX8-NEXT: [[ITMP15:%.*]] = load half, ptr addrspace(1) [[ITMP14]], align 2
-; GFX8-NEXT: [[ITMP16:%.*]] = call half @llvm.canonicalize.f16(half [[ITMP15]])
-; GFX8-NEXT: store half [[ITMP16]], ptr addrspace(1) [[ITMP14]], align 2
+; GFX8-NEXT: [[TMP2:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP10]], align 2
+; GFX8-NEXT: [[TMP3:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP2]])
+; GFX8-NEXT: store <2 x half> [[TMP3]], ptr addrspace(1) [[ITMP10]], align 2
; GFX8-NEXT: ret void
;
-; GFX9-LABEL: define void @canonicalize_combine_v4f16(
-; GFX9-SAME: ptr addrspace(1) [[ARG:%.*]]) {
-; GFX9-NEXT: [[BB:.*:]]
-; GFX9-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
-; GFX9-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
-; GFX9-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
-; GFX9-NEXT: [[ITMP5:%.*]] = add nuw nsw i64 [[ITMP1]], 1
-; GFX9-NEXT: [[ITMP6:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP5]]
-; GFX9-NEXT: [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
-; GFX9-NEXT: [[TMP1:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP0]])
-; GFX9-NEXT: store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
-; GFX9-NEXT: [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
-; GFX9-NEXT: [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
-; GFX9-NEXT: [[ITMP11:%.*]] = load half, ptr addrspace(1) [[ITMP6]], align 2
-; GFX9-NEXT: [[ITMP13:%.*]] = add nuw nsw i64 [[ITMP1]], 3
-; GFX9-NEXT: [[ITMP14:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP13]]
-; GFX9-NEXT: [[ITMP15:%.*]] = load half, ptr addrspace(1) [[ITMP14]], align 2
-; GFX9-NEXT: [[TMP2:%.*]] = insertelement <2 x half> poison, half [[ITMP11]], i64 0
-; GFX9-NEXT: [[TMP3:%.*]] = insertelement <2 x half> [[TMP2]], half [[ITMP15]], i64 1
-; GFX9-NEXT: [[TMP4:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP3]])
-; GFX9-NEXT: store <2 x half> [[TMP4]], ptr addrspace(1) [[ITMP10]], align 2
-; GFX9-NEXT: ret void
-;
bb:
%tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
%tmp1 = zext i32 %tmp to i64
@@ -584,7 +547,7 @@ bb:
%tmp9 = add nuw nsw i64 %tmp1, 2
%tmp10 = getelementptr inbounds half, ptr addrspace(1) %arg, i64 %tmp9
- %tmp11 = load half, ptr addrspace(1) %tmp6, align 2
+ %tmp11 = load half, ptr addrspace(1) %tmp10, align 2
%tmp12 = call half @llvm.canonicalize.f16(half %tmp11)
store half %tmp12, ptr addrspace(1) %tmp10, align 2
>From 6d5ee36ba39a5ea986814a7245747fb2d07412e0 Mon Sep 17 00:00:00 2001
From: Pablo Reble <pablo.reble at amd.com>
Date: Tue, 8 Sep 2026 13:01:32 -0500
Subject: [PATCH 2/3] [SLP][AMDGPU] Preserve testing pre-fix non-contiguous
gather
---
.../SLPVectorizer/AMDGPU/slp-v2f16.ll | 78 +++++++++++++++++++
1 file changed, 78 insertions(+)
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
index a0b6b0e939ce1..9364d46f20555 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
@@ -559,6 +559,84 @@ bb:
ret void
}
+; Preserves the pre-fix non-contiguous gather (insertelement over idx+1/idx+3)
+; that the typo fix above would otherwise drop.
+define void @canonicalize_combine_v4f16_noncontiguous(ptr addrspace(1) %arg) {
+; GFX8-LABEL: define void @canonicalize_combine_v4f16_noncontiguous(
+; GFX8-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX8-NEXT: [[BB:.*:]]
+; GFX8-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX8-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX8-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX8-NEXT: [[ITMP5:%.*]] = add nuw nsw i64 [[ITMP1]], 1
+; GFX8-NEXT: [[ITMP6:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP5]]
+; GFX8-NEXT: [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX8-NEXT: [[TMP1:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP0]])
+; GFX8-NEXT: store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX8-NEXT: [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
+; GFX8-NEXT: [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
+; GFX8-NEXT: [[ITMP11:%.*]] = load half, ptr addrspace(1) [[ITMP6]], align 2
+; GFX8-NEXT: [[ITMP12:%.*]] = call half @llvm.canonicalize.f16(half [[ITMP11]])
+; GFX8-NEXT: store half [[ITMP12]], ptr addrspace(1) [[ITMP10]], align 2
+; GFX8-NEXT: [[ITMP13:%.*]] = add nuw nsw i64 [[ITMP1]], 3
+; GFX8-NEXT: [[ITMP14:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP13]]
+; GFX8-NEXT: [[ITMP15:%.*]] = load half, ptr addrspace(1) [[ITMP14]], align 2
+; GFX8-NEXT: [[ITMP16:%.*]] = call half @llvm.canonicalize.f16(half [[ITMP15]])
+; GFX8-NEXT: store half [[ITMP16]], ptr addrspace(1) [[ITMP14]], align 2
+; GFX8-NEXT: ret void
+;
+; GFX9-LABEL: define void @canonicalize_combine_v4f16_noncontiguous(
+; GFX9-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX9-NEXT: [[BB:.*:]]
+; GFX9-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX9-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX9-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX9-NEXT: [[ITMP5:%.*]] = add nuw nsw i64 [[ITMP1]], 1
+; GFX9-NEXT: [[ITMP6:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP5]]
+; GFX9-NEXT: [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX9-NEXT: [[TMP1:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP0]])
+; GFX9-NEXT: store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX9-NEXT: [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
+; GFX9-NEXT: [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
+; GFX9-NEXT: [[ITMP11:%.*]] = load half, ptr addrspace(1) [[ITMP6]], align 2
+; GFX9-NEXT: [[ITMP13:%.*]] = add nuw nsw i64 [[ITMP1]], 3
+; GFX9-NEXT: [[ITMP14:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP13]]
+; GFX9-NEXT: [[ITMP15:%.*]] = load half, ptr addrspace(1) [[ITMP14]], align 2
+; GFX9-NEXT: [[TMP2:%.*]] = insertelement <2 x half> poison, half [[ITMP11]], i64 0
+; GFX9-NEXT: [[TMP3:%.*]] = insertelement <2 x half> [[TMP2]], half [[ITMP15]], i64 1
+; GFX9-NEXT: [[TMP4:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP3]])
+; GFX9-NEXT: store <2 x half> [[TMP4]], ptr addrspace(1) [[ITMP10]], align 2
+; GFX9-NEXT: ret void
+;
+bb:
+ %tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
+ %tmp1 = zext i32 %tmp to i64
+
+ %tmp2 = getelementptr inbounds half, ptr addrspace(1) %arg, i64 %tmp1
+ %tmp3 = load half, ptr addrspace(1) %tmp2, align 2
+ %tmp4 = call half @llvm.canonicalize.f16(half %tmp3)
+ store half %tmp4, ptr addrspace(1) %tmp2, align 2
+
+ %tmp5 = add nuw nsw i64 %tmp1, 1
+ %tmp6 = getelementptr inbounds half, ptr addrspace(1) %arg, i64 %tmp5
+ %tmp7 = load half, ptr addrspace(1) %tmp6, align 2
+ %tmp8 = call half @llvm.canonicalize.f16(half %tmp7)
+ store half %tmp8, ptr addrspace(1) %tmp6, align 2
+
+ %tmp9 = add nuw nsw i64 %tmp1, 2
+ %tmp10 = getelementptr inbounds half, ptr addrspace(1) %arg, i64 %tmp9
+ %tmp11 = load half, ptr addrspace(1) %tmp6, align 2
+ %tmp12 = call half @llvm.canonicalize.f16(half %tmp11)
+ store half %tmp12, ptr addrspace(1) %tmp10, align 2
+
+ %tmp13 = add nuw nsw i64 %tmp1, 3
+ %tmp14 = getelementptr inbounds half, ptr addrspace(1) %arg, i64 %tmp13
+ %tmp15 = load half, ptr addrspace(1) %tmp14, align 2
+ %tmp16 = call half @llvm.canonicalize.f16(half %tmp15)
+ store half %tmp16, ptr addrspace(1) %tmp14, align 2
+ ret void
+}
+
; FIXME: Should not vectorize on gfx8
define void @minimumnum_combine_v2f16(ptr addrspace(1) %arg) {
; GCN-LABEL: define void @minimumnum_combine_v2f16(
>From 754325b7105bf3952a321fda348f325273bb17f4 Mon Sep 17 00:00:00 2001
From: Pablo Reble <pablo.reble at amd.com>
Date: Wed, 9 Sep 2026 10:47:28 -0500
Subject: [PATCH 3/3] [SLP][AMDGPU] Fix CHECK-block coverage gap in
slp-v2f16.ll for gfx90a
---
.../SLPVectorizer/AMDGPU/slp-v2f16.ll | 105 +++++++++++++++++-
1 file changed, 104 insertions(+), 1 deletion(-)
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
index 9364d46f20555..31f5314bba8c7 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
@@ -1,7 +1,7 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --prefix-filecheck-ir-name I --version 6
; RUN: opt -S -mtriple=amdgpu8.03-amd-amdhsa -passes=slp-vectorizer < %s | FileCheck -check-prefixes=GCN,GFX8 %s
; RUN: opt -S -mtriple=amdgpu9.08-amd-amdhsa -passes=slp-vectorizer < %s | FileCheck -check-prefixes=GCN,GFX9 %s
-; RUN: opt -S -mtriple=amdgpu9.0a-amd-amdhsa -passes=slp-vectorizer < %s | FileCheck -check-prefixes=GCN,GFX9 %s
+; RUN: opt -S -mtriple=amdgpu9.0a-amd-amdhsa -passes=slp-vectorizer < %s | FileCheck -check-prefixes=GCN,GFX90A %s
; RUN: opt -S -mtriple=amdgpu10.30-amd-amdhsa -passes=slp-vectorizer < %s | FileCheck -check-prefixes=GCN,GFX9 %s
; FIXME: Should not vectorize on gfx8
@@ -228,6 +228,17 @@ define void @minnum_combine_v2f16(ptr addrspace(1) %arg) {
; GFX9-NEXT: store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
; GFX9-NEXT: ret void
;
+; GFX90A-LABEL: define void @minnum_combine_v2f16(
+; GFX90A-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX90A-NEXT: [[BB:.*:]]
+; GFX90A-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX90A-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX90A-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX90A-NEXT: [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT: [[TMP1:%.*]] = call <2 x half> @llvm.minnum.v2f16(<2 x half> [[TMP0]], <2 x half> splat (half 1.000000e+00))
+; GFX90A-NEXT: store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT: ret void
+;
bb:
%tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
%tmp1 = zext i32 %tmp to i64
@@ -272,6 +283,17 @@ define void @maxnum_combine_v2f16(ptr addrspace(1) %arg) {
; GFX9-NEXT: store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
; GFX9-NEXT: ret void
;
+; GFX90A-LABEL: define void @maxnum_combine_v2f16(
+; GFX90A-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX90A-NEXT: [[BB:.*:]]
+; GFX90A-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX90A-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX90A-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX90A-NEXT: [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT: [[TMP1:%.*]] = call <2 x half> @llvm.maxnum.v2f16(<2 x half> [[TMP0]], <2 x half> splat (half 1.000000e+00))
+; GFX90A-NEXT: store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT: ret void
+;
bb:
%tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
%tmp1 = zext i32 %tmp to i64
@@ -483,6 +505,37 @@ define void @copysign_combine_v4f16(ptr addrspace(1) %arg, half %sign) {
; GFX8-NEXT: store <2 x half> [[TMP5]], ptr addrspace(1) [[ITMP10]], align 2
; GFX8-NEXT: ret void
;
+; GFX9-LABEL: define void @copysign_combine_v4f16(
+; GFX9-SAME: ptr addrspace(1) [[ARG:%.*]], half [[SIGN:%.*]]) {
+; GFX9-NEXT: [[BB:.*:]]
+; GFX9-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX9-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX9-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX9-NEXT: [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX9-NEXT: [[TMP1:%.*]] = insertelement <2 x half> poison, half [[SIGN]], i64 0
+; GFX9-NEXT: [[TMP2:%.*]] = shufflevector <2 x half> [[TMP1]], <2 x half> poison, <2 x i32> zeroinitializer
+; GFX9-NEXT: [[TMP3:%.*]] = call <2 x half> @llvm.copysign.v2f16(<2 x half> [[TMP0]], <2 x half> [[TMP2]])
+; GFX9-NEXT: store <2 x half> [[TMP3]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX9-NEXT: [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
+; GFX9-NEXT: [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
+; GFX9-NEXT: [[TMP4:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP10]], align 2
+; GFX9-NEXT: [[TMP5:%.*]] = call <2 x half> @llvm.copysign.v2f16(<2 x half> [[TMP4]], <2 x half> [[TMP2]])
+; GFX9-NEXT: store <2 x half> [[TMP5]], ptr addrspace(1) [[ITMP10]], align 2
+; GFX9-NEXT: ret void
+;
+; GFX90A-LABEL: define void @copysign_combine_v4f16(
+; GFX90A-SAME: ptr addrspace(1) [[ARG:%.*]], half [[SIGN:%.*]]) {
+; GFX90A-NEXT: [[BB:.*:]]
+; GFX90A-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX90A-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX90A-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX90A-NEXT: [[TMP0:%.*]] = load <4 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT: [[TMP1:%.*]] = insertelement <4 x half> poison, half [[SIGN]], i64 0
+; GFX90A-NEXT: [[TMP2:%.*]] = shufflevector <4 x half> [[TMP1]], <4 x half> poison, <4 x i32> zeroinitializer
+; GFX90A-NEXT: [[TMP3:%.*]] = call <4 x half> @llvm.copysign.v4f16(<4 x half> [[TMP0]], <4 x half> [[TMP2]])
+; GFX90A-NEXT: store <4 x half> [[TMP3]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT: ret void
+;
bb:
%tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
%tmp1 = zext i32 %tmp to i64
@@ -530,6 +583,33 @@ define void @canonicalize_combine_v4f16(ptr addrspace(1) %arg) {
; GFX8-NEXT: store <2 x half> [[TMP3]], ptr addrspace(1) [[ITMP10]], align 2
; GFX8-NEXT: ret void
;
+; GFX9-LABEL: define void @canonicalize_combine_v4f16(
+; GFX9-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX9-NEXT: [[BB:.*:]]
+; GFX9-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX9-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX9-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX9-NEXT: [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX9-NEXT: [[TMP1:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP0]])
+; GFX9-NEXT: store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX9-NEXT: [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
+; GFX9-NEXT: [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
+; GFX9-NEXT: [[TMP2:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP10]], align 2
+; GFX9-NEXT: [[TMP3:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP2]])
+; GFX9-NEXT: store <2 x half> [[TMP3]], ptr addrspace(1) [[ITMP10]], align 2
+; GFX9-NEXT: ret void
+;
+; GFX90A-LABEL: define void @canonicalize_combine_v4f16(
+; GFX90A-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX90A-NEXT: [[BB:.*:]]
+; GFX90A-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX90A-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX90A-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX90A-NEXT: [[TMP0:%.*]] = load <4 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT: [[TMP1:%.*]] = call <4 x half> @llvm.canonicalize.v4f16(<4 x half> [[TMP0]])
+; GFX90A-NEXT: store <4 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT: ret void
+;
bb:
%tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
%tmp1 = zext i32 %tmp to i64
@@ -608,6 +688,29 @@ define void @canonicalize_combine_v4f16_noncontiguous(ptr addrspace(1) %arg) {
; GFX9-NEXT: store <2 x half> [[TMP4]], ptr addrspace(1) [[ITMP10]], align 2
; GFX9-NEXT: ret void
;
+; GFX90A-LABEL: define void @canonicalize_combine_v4f16_noncontiguous(
+; GFX90A-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX90A-NEXT: [[BB:.*:]]
+; GFX90A-NEXT: [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX90A-NEXT: [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX90A-NEXT: [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX90A-NEXT: [[ITMP5:%.*]] = add nuw nsw i64 [[ITMP1]], 1
+; GFX90A-NEXT: [[ITMP6:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP5]]
+; GFX90A-NEXT: [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT: [[TMP1:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP0]])
+; GFX90A-NEXT: store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT: [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
+; GFX90A-NEXT: [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
+; GFX90A-NEXT: [[ITMP11:%.*]] = load half, ptr addrspace(1) [[ITMP6]], align 2
+; GFX90A-NEXT: [[ITMP13:%.*]] = add nuw nsw i64 [[ITMP1]], 3
+; GFX90A-NEXT: [[ITMP14:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP13]]
+; GFX90A-NEXT: [[ITMP15:%.*]] = load half, ptr addrspace(1) [[ITMP14]], align 2
+; GFX90A-NEXT: [[TMP2:%.*]] = insertelement <2 x half> poison, half [[ITMP11]], i64 0
+; GFX90A-NEXT: [[TMP3:%.*]] = insertelement <2 x half> [[TMP2]], half [[ITMP15]], i64 1
+; GFX90A-NEXT: [[TMP4:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP3]])
+; GFX90A-NEXT: store <2 x half> [[TMP4]], ptr addrspace(1) [[ITMP10]], align 2
+; GFX90A-NEXT: ret void
+;
bb:
%tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
%tmp1 = zext i32 %tmp to i64
More information about the llvm-commits
mailing list