[llvm] [SLP][AMDGPU] Fix load-address typo in v4f16 combine tests (PR #222112)

Pablo Reble via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 9 08:47:46 PDT 2026


https://github.com/reble updated https://github.com/llvm/llvm-project/pull/222112

>From eee4e7578461911ed850db34e9b1016795680f38 Mon Sep 17 00:00:00 2001
From: Pablo Reble <pablo.reble at amd.com>
Date: Tue, 8 Sep 2026 12:03:57 -0500
Subject: [PATCH 1/3] [SLP][AMDGPU] Fix broken load-address typo in
 slp-v2f16.ll v4f16 tests

---
 .../SLPVectorizer/AMDGPU/slp-v2f16.ll         | 81 +++++--------------
 1 file changed, 22 insertions(+), 59 deletions(-)

diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
index 498d9ffd7f594..a0b6b0e939ce1 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
@@ -465,30 +465,23 @@ bb:
 ; FIXME: Should always vectorize
 
 define void @copysign_combine_v4f16(ptr addrspace(1) %arg, half %sign) {
-; GCN-LABEL: define void @copysign_combine_v4f16(
-; GCN-SAME: ptr addrspace(1) [[ARG:%.*]], half [[SIGN:%.*]]) {
-; GCN-NEXT:  [[BB:.*:]]
-; GCN-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
-; GCN-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
-; GCN-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
-; GCN-NEXT:    [[ITMP5:%.*]] = add nuw nsw i64 [[ITMP1]], 1
-; GCN-NEXT:    [[ITMP6:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP5]]
-; GCN-NEXT:    [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
-; GCN-NEXT:    [[TMP1:%.*]] = insertelement <2 x half> poison, half [[SIGN]], i64 0
-; GCN-NEXT:    [[TMP2:%.*]] = shufflevector <2 x half> [[TMP1]], <2 x half> poison, <2 x i32> zeroinitializer
-; GCN-NEXT:    [[TMP3:%.*]] = call <2 x half> @llvm.copysign.v2f16(<2 x half> [[TMP0]], <2 x half> [[TMP2]])
-; GCN-NEXT:    store <2 x half> [[TMP3]], ptr addrspace(1) [[ITMP2]], align 2
-; GCN-NEXT:    [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
-; GCN-NEXT:    [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
-; GCN-NEXT:    [[ITMP11:%.*]] = load half, ptr addrspace(1) [[ITMP6]], align 2
-; GCN-NEXT:    [[ITMP12:%.*]] = call half @llvm.copysign.f16(half [[ITMP11]], half [[SIGN]])
-; GCN-NEXT:    store half [[ITMP12]], ptr addrspace(1) [[ITMP10]], align 2
-; GCN-NEXT:    [[ITMP13:%.*]] = add nuw nsw i64 [[ITMP1]], 3
-; GCN-NEXT:    [[ITMP14:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP13]]
-; GCN-NEXT:    [[ITMP15:%.*]] = load half, ptr addrspace(1) [[ITMP14]], align 2
-; GCN-NEXT:    [[ITMP16:%.*]] = call half @llvm.copysign.f16(half [[ITMP15]], half [[SIGN]])
-; GCN-NEXT:    store half [[ITMP16]], ptr addrspace(1) [[ITMP14]], align 2
-; GCN-NEXT:    ret void
+; GFX8-LABEL: define void @copysign_combine_v4f16(
+; GFX8-SAME: ptr addrspace(1) [[ARG:%.*]], half [[SIGN:%.*]]) {
+; GFX8-NEXT:  [[BB:.*:]]
+; GFX8-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX8-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX8-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX8-NEXT:    [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX8-NEXT:    [[TMP1:%.*]] = insertelement <2 x half> poison, half [[SIGN]], i64 0
+; GFX8-NEXT:    [[TMP2:%.*]] = shufflevector <2 x half> [[TMP1]], <2 x half> poison, <2 x i32> zeroinitializer
+; GFX8-NEXT:    [[TMP3:%.*]] = call <2 x half> @llvm.copysign.v2f16(<2 x half> [[TMP0]], <2 x half> [[TMP2]])
+; GFX8-NEXT:    store <2 x half> [[TMP3]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX8-NEXT:    [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
+; GFX8-NEXT:    [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
+; GFX8-NEXT:    [[TMP4:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP10]], align 2
+; GFX8-NEXT:    [[TMP5:%.*]] = call <2 x half> @llvm.copysign.v2f16(<2 x half> [[TMP4]], <2 x half> [[TMP2]])
+; GFX8-NEXT:    store <2 x half> [[TMP5]], ptr addrspace(1) [[ITMP10]], align 2
+; GFX8-NEXT:    ret void
 ;
 bb:
   %tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
@@ -507,7 +500,7 @@ bb:
 
   %tmp9 = add nuw nsw i64 %tmp1, 2
   %tmp10 = getelementptr inbounds half, ptr addrspace(1) %arg, i64 %tmp9
-  %tmp11 = load half, ptr addrspace(1) %tmp6, align 2
+  %tmp11 = load half, ptr addrspace(1) %tmp10, align 2
   %tmp12 = call half @llvm.copysign.f16(half %tmp11, half %sign)
   store half %tmp12, ptr addrspace(1) %tmp10, align 2
 
@@ -527,46 +520,16 @@ define void @canonicalize_combine_v4f16(ptr addrspace(1) %arg) {
 ; GFX8-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
 ; GFX8-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
 ; GFX8-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
-; GFX8-NEXT:    [[ITMP5:%.*]] = add nuw nsw i64 [[ITMP1]], 1
-; GFX8-NEXT:    [[ITMP6:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP5]]
 ; GFX8-NEXT:    [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
 ; GFX8-NEXT:    [[TMP1:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP0]])
 ; GFX8-NEXT:    store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
 ; GFX8-NEXT:    [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
 ; GFX8-NEXT:    [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
-; GFX8-NEXT:    [[ITMP11:%.*]] = load half, ptr addrspace(1) [[ITMP6]], align 2
-; GFX8-NEXT:    [[ITMP12:%.*]] = call half @llvm.canonicalize.f16(half [[ITMP11]])
-; GFX8-NEXT:    store half [[ITMP12]], ptr addrspace(1) [[ITMP10]], align 2
-; GFX8-NEXT:    [[ITMP13:%.*]] = add nuw nsw i64 [[ITMP1]], 3
-; GFX8-NEXT:    [[ITMP14:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP13]]
-; GFX8-NEXT:    [[ITMP15:%.*]] = load half, ptr addrspace(1) [[ITMP14]], align 2
-; GFX8-NEXT:    [[ITMP16:%.*]] = call half @llvm.canonicalize.f16(half [[ITMP15]])
-; GFX8-NEXT:    store half [[ITMP16]], ptr addrspace(1) [[ITMP14]], align 2
+; GFX8-NEXT:    [[TMP2:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP10]], align 2
+; GFX8-NEXT:    [[TMP3:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP2]])
+; GFX8-NEXT:    store <2 x half> [[TMP3]], ptr addrspace(1) [[ITMP10]], align 2
 ; GFX8-NEXT:    ret void
 ;
-; GFX9-LABEL: define void @canonicalize_combine_v4f16(
-; GFX9-SAME: ptr addrspace(1) [[ARG:%.*]]) {
-; GFX9-NEXT:  [[BB:.*:]]
-; GFX9-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
-; GFX9-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
-; GFX9-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
-; GFX9-NEXT:    [[ITMP5:%.*]] = add nuw nsw i64 [[ITMP1]], 1
-; GFX9-NEXT:    [[ITMP6:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP5]]
-; GFX9-NEXT:    [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
-; GFX9-NEXT:    [[TMP1:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP0]])
-; GFX9-NEXT:    store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
-; GFX9-NEXT:    [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
-; GFX9-NEXT:    [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
-; GFX9-NEXT:    [[ITMP11:%.*]] = load half, ptr addrspace(1) [[ITMP6]], align 2
-; GFX9-NEXT:    [[ITMP13:%.*]] = add nuw nsw i64 [[ITMP1]], 3
-; GFX9-NEXT:    [[ITMP14:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP13]]
-; GFX9-NEXT:    [[ITMP15:%.*]] = load half, ptr addrspace(1) [[ITMP14]], align 2
-; GFX9-NEXT:    [[TMP2:%.*]] = insertelement <2 x half> poison, half [[ITMP11]], i64 0
-; GFX9-NEXT:    [[TMP3:%.*]] = insertelement <2 x half> [[TMP2]], half [[ITMP15]], i64 1
-; GFX9-NEXT:    [[TMP4:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP3]])
-; GFX9-NEXT:    store <2 x half> [[TMP4]], ptr addrspace(1) [[ITMP10]], align 2
-; GFX9-NEXT:    ret void
-;
 bb:
   %tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
   %tmp1 = zext i32 %tmp to i64
@@ -584,7 +547,7 @@ bb:
 
   %tmp9 = add nuw nsw i64 %tmp1, 2
   %tmp10 = getelementptr inbounds half, ptr addrspace(1) %arg, i64 %tmp9
-  %tmp11 = load half, ptr addrspace(1) %tmp6, align 2
+  %tmp11 = load half, ptr addrspace(1) %tmp10, align 2
   %tmp12 = call half @llvm.canonicalize.f16(half %tmp11)
   store half %tmp12, ptr addrspace(1) %tmp10, align 2
 

>From 6d5ee36ba39a5ea986814a7245747fb2d07412e0 Mon Sep 17 00:00:00 2001
From: Pablo Reble <pablo.reble at amd.com>
Date: Tue, 8 Sep 2026 13:01:32 -0500
Subject: [PATCH 2/3] [SLP][AMDGPU] Preserve testing pre-fix non-contiguous
 gather

---
 .../SLPVectorizer/AMDGPU/slp-v2f16.ll         | 78 +++++++++++++++++++
 1 file changed, 78 insertions(+)

diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
index a0b6b0e939ce1..9364d46f20555 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
@@ -559,6 +559,84 @@ bb:
   ret void
 }
 
+; Preserves the pre-fix non-contiguous gather (insertelement over idx+1/idx+3)
+; that the typo fix above would otherwise drop.
+define void @canonicalize_combine_v4f16_noncontiguous(ptr addrspace(1) %arg) {
+; GFX8-LABEL: define void @canonicalize_combine_v4f16_noncontiguous(
+; GFX8-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX8-NEXT:  [[BB:.*:]]
+; GFX8-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX8-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX8-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX8-NEXT:    [[ITMP5:%.*]] = add nuw nsw i64 [[ITMP1]], 1
+; GFX8-NEXT:    [[ITMP6:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP5]]
+; GFX8-NEXT:    [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX8-NEXT:    [[TMP1:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP0]])
+; GFX8-NEXT:    store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX8-NEXT:    [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
+; GFX8-NEXT:    [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
+; GFX8-NEXT:    [[ITMP11:%.*]] = load half, ptr addrspace(1) [[ITMP6]], align 2
+; GFX8-NEXT:    [[ITMP12:%.*]] = call half @llvm.canonicalize.f16(half [[ITMP11]])
+; GFX8-NEXT:    store half [[ITMP12]], ptr addrspace(1) [[ITMP10]], align 2
+; GFX8-NEXT:    [[ITMP13:%.*]] = add nuw nsw i64 [[ITMP1]], 3
+; GFX8-NEXT:    [[ITMP14:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP13]]
+; GFX8-NEXT:    [[ITMP15:%.*]] = load half, ptr addrspace(1) [[ITMP14]], align 2
+; GFX8-NEXT:    [[ITMP16:%.*]] = call half @llvm.canonicalize.f16(half [[ITMP15]])
+; GFX8-NEXT:    store half [[ITMP16]], ptr addrspace(1) [[ITMP14]], align 2
+; GFX8-NEXT:    ret void
+;
+; GFX9-LABEL: define void @canonicalize_combine_v4f16_noncontiguous(
+; GFX9-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX9-NEXT:  [[BB:.*:]]
+; GFX9-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX9-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX9-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX9-NEXT:    [[ITMP5:%.*]] = add nuw nsw i64 [[ITMP1]], 1
+; GFX9-NEXT:    [[ITMP6:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP5]]
+; GFX9-NEXT:    [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX9-NEXT:    [[TMP1:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP0]])
+; GFX9-NEXT:    store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX9-NEXT:    [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
+; GFX9-NEXT:    [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
+; GFX9-NEXT:    [[ITMP11:%.*]] = load half, ptr addrspace(1) [[ITMP6]], align 2
+; GFX9-NEXT:    [[ITMP13:%.*]] = add nuw nsw i64 [[ITMP1]], 3
+; GFX9-NEXT:    [[ITMP14:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP13]]
+; GFX9-NEXT:    [[ITMP15:%.*]] = load half, ptr addrspace(1) [[ITMP14]], align 2
+; GFX9-NEXT:    [[TMP2:%.*]] = insertelement <2 x half> poison, half [[ITMP11]], i64 0
+; GFX9-NEXT:    [[TMP3:%.*]] = insertelement <2 x half> [[TMP2]], half [[ITMP15]], i64 1
+; GFX9-NEXT:    [[TMP4:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP3]])
+; GFX9-NEXT:    store <2 x half> [[TMP4]], ptr addrspace(1) [[ITMP10]], align 2
+; GFX9-NEXT:    ret void
+;
+bb:
+  %tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
+  %tmp1 = zext i32 %tmp to i64
+
+  %tmp2 = getelementptr inbounds half, ptr addrspace(1) %arg, i64 %tmp1
+  %tmp3 = load half, ptr addrspace(1) %tmp2, align 2
+  %tmp4 = call half @llvm.canonicalize.f16(half %tmp3)
+  store half %tmp4, ptr addrspace(1) %tmp2, align 2
+
+  %tmp5 = add nuw nsw i64 %tmp1, 1
+  %tmp6 = getelementptr inbounds half, ptr addrspace(1) %arg, i64 %tmp5
+  %tmp7 = load half, ptr addrspace(1) %tmp6, align 2
+  %tmp8 = call half @llvm.canonicalize.f16(half %tmp7)
+  store half %tmp8, ptr addrspace(1) %tmp6, align 2
+
+  %tmp9 = add nuw nsw i64 %tmp1, 2
+  %tmp10 = getelementptr inbounds half, ptr addrspace(1) %arg, i64 %tmp9
+  %tmp11 = load half, ptr addrspace(1) %tmp6, align 2
+  %tmp12 = call half @llvm.canonicalize.f16(half %tmp11)
+  store half %tmp12, ptr addrspace(1) %tmp10, align 2
+
+  %tmp13 = add nuw nsw i64 %tmp1, 3
+  %tmp14 = getelementptr inbounds half, ptr addrspace(1) %arg, i64 %tmp13
+  %tmp15 = load half, ptr addrspace(1) %tmp14, align 2
+  %tmp16 = call half @llvm.canonicalize.f16(half %tmp15)
+  store half %tmp16, ptr addrspace(1) %tmp14, align 2
+  ret void
+}
+
 ; FIXME: Should not vectorize on gfx8
 define void @minimumnum_combine_v2f16(ptr addrspace(1) %arg) {
 ; GCN-LABEL: define void @minimumnum_combine_v2f16(

>From 754325b7105bf3952a321fda348f325273bb17f4 Mon Sep 17 00:00:00 2001
From: Pablo Reble <pablo.reble at amd.com>
Date: Wed, 9 Sep 2026 10:47:28 -0500
Subject: [PATCH 3/3] [SLP][AMDGPU] Fix CHECK-block coverage gap in
 slp-v2f16.ll for gfx90a

---
 .../SLPVectorizer/AMDGPU/slp-v2f16.ll         | 105 +++++++++++++++++-
 1 file changed, 104 insertions(+), 1 deletion(-)

diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
index 9364d46f20555..31f5314bba8c7 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll
@@ -1,7 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --prefix-filecheck-ir-name I --version 6
 ; RUN: opt -S -mtriple=amdgpu8.03-amd-amdhsa -passes=slp-vectorizer < %s | FileCheck -check-prefixes=GCN,GFX8 %s
 ; RUN: opt -S -mtriple=amdgpu9.08-amd-amdhsa -passes=slp-vectorizer < %s | FileCheck -check-prefixes=GCN,GFX9 %s
-; RUN: opt -S -mtriple=amdgpu9.0a-amd-amdhsa -passes=slp-vectorizer < %s | FileCheck -check-prefixes=GCN,GFX9 %s
+; RUN: opt -S -mtriple=amdgpu9.0a-amd-amdhsa -passes=slp-vectorizer < %s | FileCheck -check-prefixes=GCN,GFX90A %s
 ; RUN: opt -S -mtriple=amdgpu10.30-amd-amdhsa -passes=slp-vectorizer < %s | FileCheck -check-prefixes=GCN,GFX9 %s
 
 ; FIXME: Should not vectorize on gfx8
@@ -228,6 +228,17 @@ define void @minnum_combine_v2f16(ptr addrspace(1) %arg) {
 ; GFX9-NEXT:    store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
 ; GFX9-NEXT:    ret void
 ;
+; GFX90A-LABEL: define void @minnum_combine_v2f16(
+; GFX90A-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX90A-NEXT:  [[BB:.*:]]
+; GFX90A-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX90A-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX90A-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX90A-NEXT:    [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT:    [[TMP1:%.*]] = call <2 x half> @llvm.minnum.v2f16(<2 x half> [[TMP0]], <2 x half> splat (half 1.000000e+00))
+; GFX90A-NEXT:    store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT:    ret void
+;
 bb:
   %tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
   %tmp1 = zext i32 %tmp to i64
@@ -272,6 +283,17 @@ define void @maxnum_combine_v2f16(ptr addrspace(1) %arg) {
 ; GFX9-NEXT:    store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
 ; GFX9-NEXT:    ret void
 ;
+; GFX90A-LABEL: define void @maxnum_combine_v2f16(
+; GFX90A-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX90A-NEXT:  [[BB:.*:]]
+; GFX90A-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX90A-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX90A-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX90A-NEXT:    [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT:    [[TMP1:%.*]] = call <2 x half> @llvm.maxnum.v2f16(<2 x half> [[TMP0]], <2 x half> splat (half 1.000000e+00))
+; GFX90A-NEXT:    store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT:    ret void
+;
 bb:
   %tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
   %tmp1 = zext i32 %tmp to i64
@@ -483,6 +505,37 @@ define void @copysign_combine_v4f16(ptr addrspace(1) %arg, half %sign) {
 ; GFX8-NEXT:    store <2 x half> [[TMP5]], ptr addrspace(1) [[ITMP10]], align 2
 ; GFX8-NEXT:    ret void
 ;
+; GFX9-LABEL: define void @copysign_combine_v4f16(
+; GFX9-SAME: ptr addrspace(1) [[ARG:%.*]], half [[SIGN:%.*]]) {
+; GFX9-NEXT:  [[BB:.*:]]
+; GFX9-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX9-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX9-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX9-NEXT:    [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX9-NEXT:    [[TMP1:%.*]] = insertelement <2 x half> poison, half [[SIGN]], i64 0
+; GFX9-NEXT:    [[TMP2:%.*]] = shufflevector <2 x half> [[TMP1]], <2 x half> poison, <2 x i32> zeroinitializer
+; GFX9-NEXT:    [[TMP3:%.*]] = call <2 x half> @llvm.copysign.v2f16(<2 x half> [[TMP0]], <2 x half> [[TMP2]])
+; GFX9-NEXT:    store <2 x half> [[TMP3]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX9-NEXT:    [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
+; GFX9-NEXT:    [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
+; GFX9-NEXT:    [[TMP4:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP10]], align 2
+; GFX9-NEXT:    [[TMP5:%.*]] = call <2 x half> @llvm.copysign.v2f16(<2 x half> [[TMP4]], <2 x half> [[TMP2]])
+; GFX9-NEXT:    store <2 x half> [[TMP5]], ptr addrspace(1) [[ITMP10]], align 2
+; GFX9-NEXT:    ret void
+;
+; GFX90A-LABEL: define void @copysign_combine_v4f16(
+; GFX90A-SAME: ptr addrspace(1) [[ARG:%.*]], half [[SIGN:%.*]]) {
+; GFX90A-NEXT:  [[BB:.*:]]
+; GFX90A-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX90A-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX90A-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX90A-NEXT:    [[TMP0:%.*]] = load <4 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT:    [[TMP1:%.*]] = insertelement <4 x half> poison, half [[SIGN]], i64 0
+; GFX90A-NEXT:    [[TMP2:%.*]] = shufflevector <4 x half> [[TMP1]], <4 x half> poison, <4 x i32> zeroinitializer
+; GFX90A-NEXT:    [[TMP3:%.*]] = call <4 x half> @llvm.copysign.v4f16(<4 x half> [[TMP0]], <4 x half> [[TMP2]])
+; GFX90A-NEXT:    store <4 x half> [[TMP3]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT:    ret void
+;
 bb:
   %tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
   %tmp1 = zext i32 %tmp to i64
@@ -530,6 +583,33 @@ define void @canonicalize_combine_v4f16(ptr addrspace(1) %arg) {
 ; GFX8-NEXT:    store <2 x half> [[TMP3]], ptr addrspace(1) [[ITMP10]], align 2
 ; GFX8-NEXT:    ret void
 ;
+; GFX9-LABEL: define void @canonicalize_combine_v4f16(
+; GFX9-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX9-NEXT:  [[BB:.*:]]
+; GFX9-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX9-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX9-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX9-NEXT:    [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX9-NEXT:    [[TMP1:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP0]])
+; GFX9-NEXT:    store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX9-NEXT:    [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
+; GFX9-NEXT:    [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
+; GFX9-NEXT:    [[TMP2:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP10]], align 2
+; GFX9-NEXT:    [[TMP3:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP2]])
+; GFX9-NEXT:    store <2 x half> [[TMP3]], ptr addrspace(1) [[ITMP10]], align 2
+; GFX9-NEXT:    ret void
+;
+; GFX90A-LABEL: define void @canonicalize_combine_v4f16(
+; GFX90A-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX90A-NEXT:  [[BB:.*:]]
+; GFX90A-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX90A-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX90A-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX90A-NEXT:    [[TMP0:%.*]] = load <4 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT:    [[TMP1:%.*]] = call <4 x half> @llvm.canonicalize.v4f16(<4 x half> [[TMP0]])
+; GFX90A-NEXT:    store <4 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT:    ret void
+;
 bb:
   %tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
   %tmp1 = zext i32 %tmp to i64
@@ -608,6 +688,29 @@ define void @canonicalize_combine_v4f16_noncontiguous(ptr addrspace(1) %arg) {
 ; GFX9-NEXT:    store <2 x half> [[TMP4]], ptr addrspace(1) [[ITMP10]], align 2
 ; GFX9-NEXT:    ret void
 ;
+; GFX90A-LABEL: define void @canonicalize_combine_v4f16_noncontiguous(
+; GFX90A-SAME: ptr addrspace(1) [[ARG:%.*]]) {
+; GFX90A-NEXT:  [[BB:.*:]]
+; GFX90A-NEXT:    [[TMP:%.*]] = tail call i32 @llvm.amdgcn.workitem.id.x()
+; GFX90A-NEXT:    [[ITMP1:%.*]] = zext i32 [[TMP]] to i64
+; GFX90A-NEXT:    [[ITMP2:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP1]]
+; GFX90A-NEXT:    [[ITMP5:%.*]] = add nuw nsw i64 [[ITMP1]], 1
+; GFX90A-NEXT:    [[ITMP6:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP5]]
+; GFX90A-NEXT:    [[TMP0:%.*]] = load <2 x half>, ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT:    [[TMP1:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP0]])
+; GFX90A-NEXT:    store <2 x half> [[TMP1]], ptr addrspace(1) [[ITMP2]], align 2
+; GFX90A-NEXT:    [[ITMP9:%.*]] = add nuw nsw i64 [[ITMP1]], 2
+; GFX90A-NEXT:    [[ITMP10:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP9]]
+; GFX90A-NEXT:    [[ITMP11:%.*]] = load half, ptr addrspace(1) [[ITMP6]], align 2
+; GFX90A-NEXT:    [[ITMP13:%.*]] = add nuw nsw i64 [[ITMP1]], 3
+; GFX90A-NEXT:    [[ITMP14:%.*]] = getelementptr inbounds half, ptr addrspace(1) [[ARG]], i64 [[ITMP13]]
+; GFX90A-NEXT:    [[ITMP15:%.*]] = load half, ptr addrspace(1) [[ITMP14]], align 2
+; GFX90A-NEXT:    [[TMP2:%.*]] = insertelement <2 x half> poison, half [[ITMP11]], i64 0
+; GFX90A-NEXT:    [[TMP3:%.*]] = insertelement <2 x half> [[TMP2]], half [[ITMP15]], i64 1
+; GFX90A-NEXT:    [[TMP4:%.*]] = call <2 x half> @llvm.canonicalize.v2f16(<2 x half> [[TMP3]])
+; GFX90A-NEXT:    store <2 x half> [[TMP4]], ptr addrspace(1) [[ITMP10]], align 2
+; GFX90A-NEXT:    ret void
+;
 bb:
   %tmp = tail call i32 @llvm.amdgcn.workitem.id.x()
   %tmp1 = zext i32 %tmp to i64



More information about the llvm-commits mailing list