[llvm] [AMDGPU][TTI] Refine gfx9 packed FP32 SLP costs for pair formation and shuffles (PR #208572)
Akash Dutta via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 10:47:25 PDT 2026
https://github.com/akadutta updated https://github.com/llvm/llvm-project/pull/208572
>From bbff27ee986cd65616503872a6eb2b08296b84f4 Mon Sep 17 00:00:00 2001
From: Akash Dutta <Akash.Dutta at amd.com>
Date: Thu, 9 Jul 2026 22:04:21 +0000
Subject: [PATCH 01/11] Refine gfx9 packed FP32 SLP costs for pair formation
and shuffles
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 41 +++++++++
.../test/Analysis/CostModel/AMDGPU/maximum.ll | 85 ++++++++++++++-----
llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll | 74 +++++++++++-----
.../test/Analysis/CostModel/AMDGPU/minimum.ll | 85 ++++++++++++++-----
llvm/test/Analysis/CostModel/AMDGPU/minnum.ll | 74 +++++++++++-----
5 files changed, 273 insertions(+), 86 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 5125735faddd6..b98d83696cf46 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1042,6 +1042,32 @@ InstructionCost GCNTTIImpl::getVectorInstrCost(
VIC);
}
+ // Building a packed <2 x float> for a v_pk_*_f32 source is not always free:
+ // the two lanes must occupy an aligned VGPR pair, and the cost of an insert
+ // depends on where the inserted lane comes from.
+ // - A lane fed directly by a load is free: the load result can be
+ // allocated straight into its pair slot, with no alignment move.
+ // - A lane manufactured from compute is taxed: it typically needs a
+ // v_mov_b32 to align it into the pair. Charge 1 in TTI as a minimal
+ // non-zero cost for that alignment move; a higher per-insert tax
+ // over-penalizes SLP gather for this pattern.
+ // Taxing only the manufactured case keeps the SLP vectorizer honest about
+ // assembling pairs from non-adjacent scalars - without it SLP
+ // over-vectorizes and inflates register pressure - while leaving a genuine
+ // load-fed <2 x float> reduction free to pack.
+ //
+ // Restricted to f32: at 32-bit width the only packed VOP3P ALU ops are
+ // v_pk_{add,mul,fma}_f32 - there is no packed 32-bit integer op - so a
+ // <2 x i32> has no pair-alignment consumer and must not be taxed. Limited
+ // to gfx9 targets that expose packed FP32 (gfx90a, gfx94x, gfx950) via
+ // hasPackedFP32Ops(); gfx12+ is left unchanged pending separate evaluation.
+ if (Opcode == Instruction::InsertElement && EltSize == 32 &&
+ ST->hasPackedFP32Ops() &&
+ ST->getGeneration() == AMDGPUSubtarget::GFX9)
+ if (auto *VecTy = dyn_cast<FixedVectorType>(ValTy))
+ if (VecTy->getNumElements() == 2 && VecTy->getElementType()->isFloatTy())
+ return (Op1 && isa<LoadInst>(Op1)) ? 0 : 1;
+
// Extracts are just reads of a subregister, so are free. Inserts are
// considered free because we don't want to have any cost for scalarizing
// operations, and we don't have to copy into a different register class.
@@ -1344,6 +1370,21 @@ InstructionCost GCNTTIImpl::getShuffleCost(TTI::ShuffleKind Kind,
Kind = improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTp);
unsigned ScalarSize = DL.getTypeSizeInBits(SrcTy->getElementType());
+
+ // Packed FP32 on gfx9: keep shuffle-level costing free. The insert cost above
+ // already taxes manufactured <2 x float> lanes, and an additional per-lane
+ // shuffle tax stacks on top of that and over-penalizes profitable SLP trees.
+ // The 16/8-bit branch below relies on subword packing (multiple elements per
+ // VGPR) and does not apply to FP32, so FP32 is handled separately here.
+ //
+ // f32-only and gfx9 packed-FP32 targets only, for the same reasons as in
+ // getVectorInstrCost; gfx12+ is left unchanged pending separate evaluation.
+ if (ScalarSize == 32 && SrcTy->getElementType()->isFloatTy() &&
+ ST->hasPackedFP32Ops() &&
+ ST->getGeneration() == AMDGPUSubtarget::GFX9) {
+ return 0;
+ }
+
if (ST->getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS &&
(ScalarSize == 16 || ScalarSize == 8)) {
// Larger vector widths may require additional instructions, but are
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/maximum.ll b/llvm/test/Analysis/CostModel/AMDGPU/maximum.ll
index 0cbc395933efb..f2ba0b68d220e 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/maximum.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/maximum.ll
@@ -5,7 +5,7 @@
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=ALL,SLOWF64 %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx950 -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX950-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx90a -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX90A-SIZE %s
-; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx900 -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx900 -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX900-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,SLOW-SIZE %s
define void @maximum_f16() {
@@ -155,30 +155,75 @@ define void @maximum_bf16() {
define void @maximum_f32() {
; GFX950-FASTF64-LABEL: 'maximum_f32'
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-LABEL: 'maximum_f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+; GFX90A-FASTF64-LABEL: 'maximum_f32'
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 22 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; SIZE-LABEL: 'maximum_f32'
-; SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; FASTF64-LABEL: 'maximum_f32'
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; SLOWF64-LABEL: 'maximum_f32'
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; GFX950-SIZE-LABEL: 'maximum_f32'
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; GFX90A-SIZE-LABEL: 'maximum_f32'
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; GFX900-SIZE-LABEL: 'maximum_f32'
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; SLOW-SIZE-LABEL: 'maximum_f32'
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%f32 = call float @llvm.maximum.f32(float undef, float undef)
%v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
@@ -225,7 +270,3 @@ define void @maximum_f64() {
%v16f64 = call <16 x double> @llvm.maximum.v16f64(<16 x double> undef, <16 x double> undef)
ret void
}
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; FASTF64: {{.*}}
-; GFX90A-FASTF64: {{.*}}
-; GFX90A-SIZE: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll b/llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll
index 0d423fec7adcc..c1d819ead76c7 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll
@@ -3,7 +3,7 @@
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx900 -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=ALL,GFX9,FASTF64 %s
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=ALL,SLOWF64 %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx90a -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX90A-SIZE %s
-; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx900 -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx900 -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX900-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,SLOW-SIZE %s
define void @maxnum_f16() {
@@ -115,23 +115,59 @@ define void @maxnum_bf16() {
}
define void @maxnum_f32() {
-; ALL-LABEL: 'maxnum_f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+; GFX90A-FASTF64-LABEL: 'maxnum_f32'
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; SIZE-LABEL: 'maxnum_f32'
-; SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; FASTF64-LABEL: 'maxnum_f32'
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; SLOWF64-LABEL: 'maxnum_f32'
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; GFX90A-SIZE-LABEL: 'maxnum_f32'
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; GFX900-SIZE-LABEL: 'maxnum_f32'
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; SLOW-SIZE-LABEL: 'maxnum_f32'
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%f32 = call float @llvm.maxnum.f32(float undef, float undef)
%v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
@@ -169,7 +205,3 @@ define void @maxnum_f64() {
%v16f64 = call <16 x double> @llvm.maxnum.v16f64(<16 x double> undef, <16 x double> undef)
ret void
}
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; FASTF64: {{.*}}
-; GFX90A-FASTF64: {{.*}}
-; GFX90A-SIZE: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/minimum.ll b/llvm/test/Analysis/CostModel/AMDGPU/minimum.ll
index 64520379e6d55..84af2a7b1fb79 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/minimum.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/minimum.ll
@@ -5,7 +5,7 @@
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=ALL,SLOWF64 %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx950 -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX950-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx90a -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX90A-SIZE %s
-; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx900 -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx900 -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX900-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,SLOW-SIZE %s
define void @minimum_f16() {
@@ -155,30 +155,75 @@ define void @minimum_bf16() {
define void @minimum_f32() {
; GFX950-FASTF64-LABEL: 'minimum_f32'
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-LABEL: 'minimum_f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+; GFX90A-FASTF64-LABEL: 'minimum_f32'
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 22 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; SIZE-LABEL: 'minimum_f32'
-; SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; FASTF64-LABEL: 'minimum_f32'
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; SLOWF64-LABEL: 'minimum_f32'
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; GFX950-SIZE-LABEL: 'minimum_f32'
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; GFX90A-SIZE-LABEL: 'minimum_f32'
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; GFX900-SIZE-LABEL: 'minimum_f32'
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; SLOW-SIZE-LABEL: 'minimum_f32'
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%f32 = call float @llvm.minimum.f32(float undef, float undef)
%v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
@@ -225,7 +270,3 @@ define void @minimum_f64() {
%v16f64 = call <16 x double> @llvm.minimum.v16f64(<16 x double> undef, <16 x double> undef)
ret void
}
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; FASTF64: {{.*}}
-; GFX90A-FASTF64: {{.*}}
-; GFX90A-SIZE: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/minnum.ll b/llvm/test/Analysis/CostModel/AMDGPU/minnum.ll
index 61432ba440920..f0aeab214b587 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/minnum.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/minnum.ll
@@ -3,7 +3,7 @@
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx900 -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=ALL,GFX9,FASTF64 %s
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=ALL,SLOWF64 %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx90a -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX90A-SIZE %s
-; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx900 -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mcpu=gfx900 -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX900-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,SLOW-SIZE %s
define void @minnum_f16() {
@@ -115,23 +115,59 @@ define void @minnum_bf16() {
}
define void @minnum_f32() {
-; ALL-LABEL: 'minnum_f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+; GFX90A-FASTF64-LABEL: 'minnum_f32'
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; SIZE-LABEL: 'minnum_f32'
-; SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
-; SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; FASTF64-LABEL: 'minnum_f32'
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
+; FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; SLOWF64-LABEL: 'minnum_f32'
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; GFX90A-SIZE-LABEL: 'minnum_f32'
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; GFX900-SIZE-LABEL: 'minnum_f32'
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; SLOW-SIZE-LABEL: 'minnum_f32'
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%f32 = call float @llvm.minnum.f32(float undef, float undef)
%v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
@@ -169,7 +205,3 @@ define void @minnum_f64() {
%v16f64 = call <16 x double> @llvm.minnum.v16f64(<16 x double> undef, <16 x double> undef)
ret void
}
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; FASTF64: {{.*}}
-; GFX90A-FASTF64: {{.*}}
-; GFX90A-SIZE: {{.*}}
>From 0e8c3d7f24863607b313b65de53cb0bbc2dbe26a Mon Sep 17 00:00:00 2001
From: Akash Dutta <Akash.Dutta at amd.com>
Date: Thu, 9 Jul 2026 22:28:15 +0000
Subject: [PATCH 02/11] format fixes
---
llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp | 9 ++++-----
1 file changed, 4 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index b98d83696cf46..13000508b59ad 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1062,10 +1062,10 @@ InstructionCost GCNTTIImpl::getVectorInstrCost(
// to gfx9 targets that expose packed FP32 (gfx90a, gfx94x, gfx950) via
// hasPackedFP32Ops(); gfx12+ is left unchanged pending separate evaluation.
if (Opcode == Instruction::InsertElement && EltSize == 32 &&
- ST->hasPackedFP32Ops() &&
- ST->getGeneration() == AMDGPUSubtarget::GFX9)
+ ST->hasPackedFP32Ops() && ST->getGeneration() == AMDGPUSubtarget::GFX9)
if (auto *VecTy = dyn_cast<FixedVectorType>(ValTy))
- if (VecTy->getNumElements() == 2 && VecTy->getElementType()->isFloatTy())
+ if (VecTy->getNumElements() == 2 &&
+ VecTy->getElementType()->isFloatTy())
return (Op1 && isa<LoadInst>(Op1)) ? 0 : 1;
// Extracts are just reads of a subregister, so are free. Inserts are
@@ -1380,8 +1380,7 @@ InstructionCost GCNTTIImpl::getShuffleCost(TTI::ShuffleKind Kind,
// f32-only and gfx9 packed-FP32 targets only, for the same reasons as in
// getVectorInstrCost; gfx12+ is left unchanged pending separate evaluation.
if (ScalarSize == 32 && SrcTy->getElementType()->isFloatTy() &&
- ST->hasPackedFP32Ops() &&
- ST->getGeneration() == AMDGPUSubtarget::GFX9) {
+ ST->hasPackedFP32Ops() && ST->getGeneration() == AMDGPUSubtarget::GFX9) {
return 0;
}
>From ee749571cdd923018e421b4914a90a6a60c98a50 Mon Sep 17 00:00:00 2001
From: Akash Dutta <Akash.Dutta at amd.com>
Date: Fri, 10 Jul 2026 12:30:19 +0000
Subject: [PATCH 03/11] reduce comment size
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 34 ++++---------------
1 file changed, 7 insertions(+), 27 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 13000508b59ad..d4d835513d81f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1042,25 +1042,10 @@ InstructionCost GCNTTIImpl::getVectorInstrCost(
VIC);
}
- // Building a packed <2 x float> for a v_pk_*_f32 source is not always free:
- // the two lanes must occupy an aligned VGPR pair, and the cost of an insert
- // depends on where the inserted lane comes from.
- // - A lane fed directly by a load is free: the load result can be
- // allocated straight into its pair slot, with no alignment move.
- // - A lane manufactured from compute is taxed: it typically needs a
- // v_mov_b32 to align it into the pair. Charge 1 in TTI as a minimal
- // non-zero cost for that alignment move; a higher per-insert tax
- // over-penalizes SLP gather for this pattern.
- // Taxing only the manufactured case keeps the SLP vectorizer honest about
- // assembling pairs from non-adjacent scalars - without it SLP
- // over-vectorizes and inflates register pressure - while leaving a genuine
- // load-fed <2 x float> reduction free to pack.
- //
- // Restricted to f32: at 32-bit width the only packed VOP3P ALU ops are
- // v_pk_{add,mul,fma}_f32 - there is no packed 32-bit integer op - so a
- // <2 x i32> has no pair-alignment consumer and must not be taxed. Limited
- // to gfx9 targets that expose packed FP32 (gfx90a, gfx94x, gfx950) via
- // hasPackedFP32Ops(); gfx12+ is left unchanged pending separate evaluation.
+ // Gfx9 packed <2 x f32> pair formation for v_pk_*_f32: lanes must occupy an
+ // aligned VGPR pair. A load-fed insert can be allocated into its slot for
+ // free; a compute-fed insert typically needs an alignment move, so charge 1.
+ // f32-only on gfx9 targets with packed FP32 ops.
if (Opcode == Instruction::InsertElement && EltSize == 32 &&
ST->hasPackedFP32Ops() && ST->getGeneration() == AMDGPUSubtarget::GFX9)
if (auto *VecTy = dyn_cast<FixedVectorType>(ValTy))
@@ -1371,14 +1356,9 @@ InstructionCost GCNTTIImpl::getShuffleCost(TTI::ShuffleKind Kind,
unsigned ScalarSize = DL.getTypeSizeInBits(SrcTy->getElementType());
- // Packed FP32 on gfx9: keep shuffle-level costing free. The insert cost above
- // already taxes manufactured <2 x float> lanes, and an additional per-lane
- // shuffle tax stacks on top of that and over-penalizes profitable SLP trees.
- // The 16/8-bit branch below relies on subword packing (multiple elements per
- // VGPR) and does not apply to FP32, so FP32 is handled separately here.
- //
- // f32-only and gfx9 packed-FP32 targets only, for the same reasons as in
- // getVectorInstrCost; gfx12+ is left unchanged pending separate evaluation.
+ // Gfx9 packed FP32 shuffles are free. InsertElement above already taxes
+ // assembling <2 x f32> pairs, and a per-lane shuffle cost stacks on top and
+ // over-penalizes SLP. f32-only on gfx9 targets with packed FP32 ops.
if (ScalarSize == 32 && SrcTy->getElementType()->isFloatTy() &&
ST->hasPackedFP32Ops() && ST->getGeneration() == AMDGPUSubtarget::GFX9) {
return 0;
>From afb347b7f695a21f83f87656d4bc9260a5bbd2ad Mon Sep 17 00:00:00 2001
From: Akash Dutta <Akash.Dutta at amd.com>
Date: Fri, 17 Jul 2026 19:39:43 +0000
Subject: [PATCH 04/11] remove GFX9 check
---
llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp | 6 +++---
1 file changed, 3 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index d4d835513d81f..7dcaae61cb498 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1044,10 +1044,10 @@ InstructionCost GCNTTIImpl::getVectorInstrCost(
// Gfx9 packed <2 x f32> pair formation for v_pk_*_f32: lanes must occupy an
// aligned VGPR pair. A load-fed insert can be allocated into its slot for
- // free; a compute-fed insert typically needs an alignment move, so charge 1.
- // f32-only on gfx9 targets with packed FP32 ops.
+ // free; a compute-fed insert typically needs an alignment move, so
+ // charge 1. f32-only on gfx9 targets with packed FP32 ops.
if (Opcode == Instruction::InsertElement && EltSize == 32 &&
- ST->hasPackedFP32Ops() && ST->getGeneration() == AMDGPUSubtarget::GFX9)
+ ST->hasPackedFP32Ops())
if (auto *VecTy = dyn_cast<FixedVectorType>(ValTy))
if (VecTy->getNumElements() == 2 &&
VecTy->getElementType()->isFloatTy())
>From befe07d7f7e65e1d7f9eb12e06a5581d259deddd Mon Sep 17 00:00:00 2001
From: Akash Dutta <Akash.Dutta at amd.com>
Date: Fri, 17 Jul 2026 20:48:06 +0000
Subject: [PATCH 05/11] use legalization cost for packed f32 InsertElement in
TTI
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 15 +-
llvm/test/Analysis/CostModel/AMDGPU/cast.ll | 72 +++----
llvm/test/Analysis/CostModel/AMDGPU/fround.ll | 191 ++++++------------
.../test/Analysis/CostModel/AMDGPU/maximum.ll | 32 +--
llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll | 16 +-
.../test/Analysis/CostModel/AMDGPU/minimum.ll | 32 +--
llvm/test/Analysis/CostModel/AMDGPU/minnum.ll | 16 +-
...otriviallyvectorizableintrinsicoperands.ll | 33 ++-
.../AMDGPU/combine-scalar-selects.ll | 43 ++--
9 files changed, 205 insertions(+), 245 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 7dcaae61cb498..b62854f1b78cf 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1042,16 +1042,17 @@ InstructionCost GCNTTIImpl::getVectorInstrCost(
VIC);
}
- // Gfx9 packed <2 x f32> pair formation for v_pk_*_f32: lanes must occupy an
- // aligned VGPR pair. A load-fed insert can be allocated into its slot for
- // free; a compute-fed insert typically needs an alignment move, so
- // charge 1. f32-only on gfx9 targets with packed FP32 ops.
+ // Gfx9 packed f32 pair formation for v_pk_*_f32: load-fed inserts are free;
+ // compute-fed inserts cost scales with legalization (wide vectors split to
+ // native <2 x f32>).
if (Opcode == Instruction::InsertElement && EltSize == 32 &&
ST->hasPackedFP32Ops())
if (auto *VecTy = dyn_cast<FixedVectorType>(ValTy))
- if (VecTy->getNumElements() == 2 &&
- VecTy->getElementType()->isFloatTy())
- return (Op1 && isa<LoadInst>(Op1)) ? 0 : 1;
+ if (VecTy->getElementType()->isFloatTy()) {
+ if (Op1 && isa<LoadInst>(Op1))
+ return 0;
+ return getTypeLegalizationCost(ValTy).first;
+ }
// Extracts are just reads of a subregister, so are free. Inserts are
// considered free because we don't want to have any cost for scalarizing
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/cast.ll b/llvm/test/Analysis/CostModel/AMDGPU/cast.ll
index 65d1eacfbb611..444e74577bc4b 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/cast.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/cast.ll
@@ -299,19 +299,19 @@ define void @sitofp4(<4 x i1> %a, <4 x i8> %b, <4 x i16> %c, <4 x i32> %d) {
}
define void @sitofp8(<8 x i1> %a, <8 x i8> %b, <8 x i16> %c, <8 x i32> %d) {
-; ALL-LABEL: 'sitofp8'
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = sitofp <8 x i1> %a to <8 x float>
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = sitofp <8 x i8> %b to <8 x float>
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = sitofp <8 x i16> %c to <8 x float>
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = sitofp <8 x i32> %d to <8 x float>
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+; SLOW-LABEL: 'sitofp8'
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = sitofp <8 x i1> %a to <8 x float>
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = sitofp <8 x i8> %b to <8 x float>
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = sitofp <8 x i16> %c to <8 x float>
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = sitofp <8 x i32> %d to <8 x float>
+; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'sitofp8'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = sitofp <8 x i1> %a to <8 x float>
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = sitofp <8 x i8> %b to <8 x float>
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = sitofp <8 x i16> %c to <8 x float>
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = sitofp <8 x i32> %d to <8 x float>
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; SLOW-SIZE-LABEL: 'sitofp8'
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = sitofp <8 x i1> %a to <8 x float>
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = sitofp <8 x i8> %b to <8 x float>
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = sitofp <8 x i16> %c to <8 x float>
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = sitofp <8 x i32> %d to <8 x float>
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%A1 = sitofp <8 x i1> %a to <8 x float>
%B1 = sitofp <8 x i8> %b to <8 x float>
@@ -377,19 +377,19 @@ define void @uitofp4(<4 x i1> %a, <4 x i8> %b, <4 x i16> %c, <4 x i32> %d) {
}
define void @uitofp8(<8 x i1> %a, <8 x i8> %b, <8 x i16> %c, <8 x i32> %d) {
-; ALL-LABEL: 'uitofp8'
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = uitofp <8 x i1> %a to <8 x float>
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = uitofp <8 x i8> %b to <8 x float>
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = uitofp <8 x i16> %c to <8 x float>
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = uitofp <8 x i32> %d to <8 x float>
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+; SLOW-LABEL: 'uitofp8'
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = uitofp <8 x i1> %a to <8 x float>
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = uitofp <8 x i8> %b to <8 x float>
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = uitofp <8 x i16> %c to <8 x float>
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = uitofp <8 x i32> %d to <8 x float>
+; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'uitofp8'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = uitofp <8 x i1> %a to <8 x float>
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = uitofp <8 x i8> %b to <8 x float>
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = uitofp <8 x i16> %c to <8 x float>
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = uitofp <8 x i32> %d to <8 x float>
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; SLOW-SIZE-LABEL: 'uitofp8'
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = uitofp <8 x i1> %a to <8 x float>
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = uitofp <8 x i8> %b to <8 x float>
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = uitofp <8 x i16> %c to <8 x float>
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = uitofp <8 x i32> %d to <8 x float>
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%A1 = uitofp <8 x i1> %a to <8 x float>
%B1 = uitofp <8 x i8> %b to <8 x float>
@@ -399,19 +399,19 @@ define void @uitofp8(<8 x i1> %a, <8 x i8> %b, <8 x i16> %c, <8 x i32> %d) {
}
define void @fp_conv(<8 x float> %a, <16 x float>%b, <4 x float> %c) {
-; ALL-LABEL: 'fp_conv'
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A1 = fpext <4 x float> %c to <4 x double>
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A2 = fpext <8 x float> %a to <8 x double>
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A3 = fptrunc <4 x double> undef to <4 x float>
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A4 = fptrunc <8 x double> undef to <8 x float>
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+; SLOW-LABEL: 'fp_conv'
+; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A1 = fpext <4 x float> %c to <4 x double>
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A2 = fpext <8 x float> %a to <8 x double>
+; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A3 = fptrunc <4 x double> undef to <4 x float>
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A4 = fptrunc <8 x double> undef to <8 x float>
+; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'fp_conv'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A1 = fpext <4 x float> %c to <4 x double>
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A2 = fpext <8 x float> %a to <8 x double>
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A3 = fptrunc <4 x double> undef to <4 x float>
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A4 = fptrunc <8 x double> undef to <8 x float>
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; SLOW-SIZE-LABEL: 'fp_conv'
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A1 = fpext <4 x float> %c to <4 x double>
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A2 = fpext <8 x float> %a to <8 x double>
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A3 = fptrunc <4 x double> undef to <4 x float>
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A4 = fptrunc <8 x double> undef to <8 x float>
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%A1 = fpext <4 x float> %c to <4 x double>
%A2 = fpext <8 x float> %a to <8 x double>
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/fround.ll b/llvm/test/Analysis/CostModel/AMDGPU/fround.ll
index 90e3ff231b5fa..d78c50d0221e4 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/fround.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/fround.ll
@@ -11,17 +11,6 @@
; END.
define i32 @ceil(i32 %arg) {
-; FAST-LABEL: 'ceil'
-; FAST-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.ceil.f32(float undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.ceil.v4f32(<4 x float> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.ceil.v8f32(<8 x float> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.ceil.v16f32(<16 x float> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.ceil.f64(double undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.ceil.v2f64(<2 x double> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.ceil.v4f64(<4 x double> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.ceil.v8f64(<8 x double> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
-;
; SLOW-LABEL: 'ceil'
; SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.ceil.f32(float undef)
; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.ceil.v4f32(<4 x float> undef)
@@ -33,17 +22,6 @@ define i32 @ceil(i32 %arg) {
; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.ceil.v8f64(<8 x double> undef)
; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
-; FAST-SIZE-LABEL: 'ceil'
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.ceil.f32(float undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.ceil.v4f32(<4 x float> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.ceil.v8f32(<8 x float> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.ceil.v16f32(<16 x float> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.ceil.f64(double undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.ceil.v2f64(<2 x double> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.ceil.v4f64(<4 x double> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.ceil.v8f64(<8 x double> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
-;
; SLOW-SIZE-LABEL: 'ceil'
; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.ceil.f32(float undef)
; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.ceil.v4f32(<4 x float> undef)
@@ -69,27 +47,27 @@ define i32 @ceil(i32 %arg) {
}
define i32 @floor(i32 %arg) {
-; ALL-LABEL: 'floor'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.floor.f32(float undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.floor.v4f32(<4 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.floor.v8f32(<8 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.floor.v16f32(<16 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.floor.f64(double undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.floor.v2f64(<2 x double> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.floor.v4f64(<4 x double> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.floor.v8f64(<8 x double> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
+; SLOW-LABEL: 'floor'
+; SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.floor.f32(float undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.floor.v4f32(<4 x float> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.floor.v8f32(<8 x float> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.floor.v16f32(<16 x float> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.floor.f64(double undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.floor.v2f64(<2 x double> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.floor.v4f64(<4 x double> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.floor.v8f64(<8 x double> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
-; ALL-SIZE-LABEL: 'floor'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.floor.f32(float undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.floor.v4f32(<4 x float> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.floor.v8f32(<8 x float> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.floor.v16f32(<16 x float> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.floor.f64(double undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.floor.v2f64(<2 x double> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.floor.v4f64(<4 x double> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.floor.v8f64(<8 x double> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
+; SLOW-SIZE-LABEL: 'floor'
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.floor.f32(float undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.floor.v4f32(<4 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.floor.v8f32(<8 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.floor.v16f32(<16 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.floor.f64(double undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.floor.v2f64(<2 x double> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.floor.v4f64(<4 x double> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.floor.v8f64(<8 x double> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
;
%F32 = call float @llvm.floor.f32(float undef)
%V4F32 = call <4 x float> @llvm.floor.v4f32(<4 x float> undef)
@@ -105,27 +83,27 @@ define i32 @floor(i32 %arg) {
}
define i32 @nearbyint(i32 %arg) {
-; ALL-LABEL: 'nearbyint'
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.nearbyint.f32(float undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.nearbyint.v4f32(<4 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.nearbyint.v8f32(<8 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.nearbyint.v16f32(<16 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.nearbyint.f64(double undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.nearbyint.v2f64(<2 x double> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.nearbyint.v4f64(<4 x double> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.nearbyint.v8f64(<8 x double> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
+; SLOW-LABEL: 'nearbyint'
+; SLOW-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.nearbyint.f32(float undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.nearbyint.v4f32(<4 x float> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.nearbyint.v8f32(<8 x float> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.nearbyint.v16f32(<16 x float> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.nearbyint.f64(double undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.nearbyint.v2f64(<2 x double> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.nearbyint.v4f64(<4 x double> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.nearbyint.v8f64(<8 x double> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
-; ALL-SIZE-LABEL: 'nearbyint'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.nearbyint.f32(float undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.nearbyint.v4f32(<4 x float> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.nearbyint.v8f32(<8 x float> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.nearbyint.v16f32(<16 x float> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.nearbyint.f64(double undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.nearbyint.v2f64(<2 x double> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.nearbyint.v4f64(<4 x double> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.nearbyint.v8f64(<8 x double> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
+; SLOW-SIZE-LABEL: 'nearbyint'
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.nearbyint.f32(float undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.nearbyint.v4f32(<4 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.nearbyint.v8f32(<8 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.nearbyint.v16f32(<16 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.nearbyint.f64(double undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.nearbyint.v2f64(<2 x double> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.nearbyint.v4f64(<4 x double> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.nearbyint.v8f64(<8 x double> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
;
%F32 = call float @llvm.nearbyint.f32(float undef)
%V4F32 = call <4 x float> @llvm.nearbyint.v4f32(<4 x float> undef)
@@ -141,27 +119,27 @@ define i32 @nearbyint(i32 %arg) {
}
define i32 @rint(i32 %arg) {
-; ALL-LABEL: 'rint'
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.rint.f32(float undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.rint.v4f32(<4 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.rint.v8f32(<8 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.rint.v16f32(<16 x float> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.rint.f64(double undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.rint.v2f64(<2 x double> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.rint.v4f64(<4 x double> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.rint.v8f64(<8 x double> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
+; SLOW-LABEL: 'rint'
+; SLOW-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.rint.f32(float undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.rint.v4f32(<4 x float> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.rint.v8f32(<8 x float> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.rint.v16f32(<16 x float> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.rint.f64(double undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.rint.v2f64(<2 x double> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.rint.v4f64(<4 x double> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.rint.v8f64(<8 x double> undef)
+; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
-; ALL-SIZE-LABEL: 'rint'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.rint.f32(float undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.rint.v4f32(<4 x float> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.rint.v8f32(<8 x float> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.rint.v16f32(<16 x float> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.rint.f64(double undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.rint.v2f64(<2 x double> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.rint.v4f64(<4 x double> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.rint.v8f64(<8 x double> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
+; SLOW-SIZE-LABEL: 'rint'
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.rint.f32(float undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.rint.v4f32(<4 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.rint.v8f32(<8 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.rint.v16f32(<16 x float> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.rint.f64(double undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.rint.v2f64(<2 x double> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.rint.v4f64(<4 x double> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.rint.v8f64(<8 x double> undef)
+; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
;
%F32 = call float @llvm.rint.f32(float undef)
%V4F32 = call <4 x float> @llvm.rint.v4f32(<4 x float> undef)
@@ -177,17 +155,6 @@ define i32 @rint(i32 %arg) {
}
define i32 @roundeven(i32 %arg) {
-; FAST-LABEL: 'roundeven'
-; FAST-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.roundeven.f32(float undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.roundeven.v4f32(<4 x float> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.roundeven.v8f32(<8 x float> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.roundeven.v16f32(<16 x float> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.roundeven.f64(double undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.roundeven.v2f64(<2 x double> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.roundeven.v4f64(<4 x double> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.roundeven.v8f64(<8 x double> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
-;
; SLOW-LABEL: 'roundeven'
; SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.roundeven.f32(float undef)
; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.roundeven.v4f32(<4 x float> undef)
@@ -199,17 +166,6 @@ define i32 @roundeven(i32 %arg) {
; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.roundeven.v8f64(<8 x double> undef)
; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
-; FAST-SIZE-LABEL: 'roundeven'
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.roundeven.f32(float undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.roundeven.v4f32(<4 x float> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.roundeven.v8f32(<8 x float> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.roundeven.v16f32(<16 x float> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.roundeven.f64(double undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.roundeven.v2f64(<2 x double> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.roundeven.v4f64(<4 x double> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.roundeven.v8f64(<8 x double> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
-;
; SLOW-SIZE-LABEL: 'roundeven'
; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.roundeven.f32(float undef)
; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.roundeven.v4f32(<4 x float> undef)
@@ -235,17 +191,6 @@ define i32 @roundeven(i32 %arg) {
}
define i32 @trunc(i32 %arg) {
-; FAST-LABEL: 'trunc'
-; FAST-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.trunc.f32(float undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.trunc.v4f32(<4 x float> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.trunc.v8f32(<8 x float> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.trunc.v16f32(<16 x float> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.trunc.f64(double undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.trunc.v2f64(<2 x double> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.trunc.v4f64(<4 x double> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.trunc.v8f64(<8 x double> undef)
-; FAST-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
-;
; SLOW-LABEL: 'trunc'
; SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.trunc.f32(float undef)
; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.trunc.v4f32(<4 x float> undef)
@@ -257,17 +202,6 @@ define i32 @trunc(i32 %arg) {
; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.trunc.v8f64(<8 x double> undef)
; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
-; FAST-SIZE-LABEL: 'trunc'
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.trunc.f32(float undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.trunc.v4f32(<4 x float> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.trunc.v8f32(<8 x float> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.trunc.v16f32(<16 x float> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.trunc.f64(double undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.trunc.v2f64(<2 x double> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.trunc.v4f64(<4 x double> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.trunc.v8f64(<8 x double> undef)
-; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
-;
; SLOW-SIZE-LABEL: 'trunc'
; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.trunc.f32(float undef)
; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.trunc.v4f32(<4 x float> undef)
@@ -351,3 +285,8 @@ declare double @llvm.trunc.f64(double)
declare <2 x double> @llvm.trunc.v2f64(<2 x double>)
declare <4 x double> @llvm.trunc.v4f64(<4 x double>)
declare <8 x double> @llvm.trunc.v8f64(<8 x double>)
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; ALL: {{.*}}
+; ALL-SIZE: {{.*}}
+; FAST: {{.*}}
+; FAST-SIZE: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/maximum.ll b/llvm/test/Analysis/CostModel/AMDGPU/maximum.ll
index f29d32cf79f84..37814fd1bf9dd 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/maximum.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/maximum.ll
@@ -156,19 +156,19 @@ define void @maximum_f32() {
; GFX950-FASTF64-LABEL: 'maximum_f32'
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 64 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
; GFX90A-FASTF64-LABEL: 'maximum_f32'
; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 22 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 44 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 88 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 208 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
; FASTF64-LABEL: 'maximum_f32'
@@ -192,19 +192,19 @@ define void @maximum_f32() {
; GFX950-SIZE-LABEL: 'maximum_f32'
; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 64 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; GFX90A-SIZE-LABEL: 'maximum_f32'
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 64 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; GFX900-SIZE-LABEL: 'maximum_f32'
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll b/llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll
index 5e0002c20bc4d..58962c0ed8cda 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll
@@ -118,10 +118,10 @@ define void @maxnum_f32() {
; GFX90A-FASTF64-LABEL: 'maxnum_f32'
; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 12 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 24 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
; FASTF64-LABEL: 'maxnum_f32'
@@ -145,10 +145,10 @@ define void @maxnum_f32() {
; GFX90A-SIZE-LABEL: 'maxnum_f32'
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 12 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 24 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; GFX900-SIZE-LABEL: 'maxnum_f32'
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/minimum.ll b/llvm/test/Analysis/CostModel/AMDGPU/minimum.ll
index 5366f3889df0a..815b9978a87ee 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/minimum.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/minimum.ll
@@ -156,19 +156,19 @@ define void @minimum_f32() {
; GFX950-FASTF64-LABEL: 'minimum_f32'
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 64 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
; GFX90A-FASTF64-LABEL: 'minimum_f32'
; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 22 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 44 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 88 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 208 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
; FASTF64-LABEL: 'minimum_f32'
@@ -192,19 +192,19 @@ define void @minimum_f32() {
; GFX950-SIZE-LABEL: 'minimum_f32'
; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 64 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; GFX90A-SIZE-LABEL: 'minimum_f32'
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 64 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; GFX900-SIZE-LABEL: 'minimum_f32'
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/minnum.ll b/llvm/test/Analysis/CostModel/AMDGPU/minnum.ll
index 140d505ede7c9..baabdfd847fbf 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/minnum.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/minnum.ll
@@ -118,10 +118,10 @@ define void @minnum_f32() {
; GFX90A-FASTF64-LABEL: 'minnum_f32'
; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 12 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 24 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
; FASTF64-LABEL: 'minnum_f32'
@@ -145,10 +145,10 @@ define void @minnum_f32() {
; GFX90A-SIZE-LABEL: 'minnum_f32'
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 12 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 24 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; GFX900-SIZE-LABEL: 'minnum_f32'
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
index 533f8bce9ed18..4678c506846ba 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
@@ -542,9 +542,17 @@ define amdgpu_kernel void @test_single_exp_hreduction(
; GCN-SAME: ptr addrspace(1) [[INPUT:%.*]], ptr addrspace(1) [[OUTPUT:%.*]]) {
; GCN-NEXT: [[ENTRY:.*:]]
; GCN-NEXT: [[P0:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 0
-; GCN-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr addrspace(1) [[P0]], align 4
-; GCN-NEXT: [[TMP1:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP0]])
-; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP1]])
+; GCN-NEXT: [[P1:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 1
+; GCN-NEXT: [[P2:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 2
+; GCN-NEXT: [[P3:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 3
+; GCN-NEXT: [[A0:%.*]] = load float, ptr addrspace(1) [[P0]], align 4
+; GCN-NEXT: [[A1:%.*]] = load float, ptr addrspace(1) [[P1]], align 4
+; GCN-NEXT: [[A2:%.*]] = load float, ptr addrspace(1) [[P2]], align 4
+; GCN-NEXT: [[A3:%.*]] = load float, ptr addrspace(1) [[P3]], align 4
+; GCN-NEXT: [[ADD01:%.*]] = fadd fast float [[A0]], [[A1]]
+; GCN-NEXT: [[ADD23:%.*]] = fadd fast float [[A2]], [[A3]]
+; GCN-NEXT: [[SUM:%.*]] = fadd fast float [[ADD01]], [[ADD23]]
+; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[SUM]])
; GCN-NEXT: store float [[EXP0]], ptr addrspace(1) [[OUTPUT]], align 4
; GCN-NEXT: ret void
;
@@ -574,13 +582,18 @@ define amdgpu_kernel void @test_hreduction_into_exp(
; GCN-SAME: ptr addrspace(1) [[INPUT:%.*]], ptr addrspace(1) [[OUTPUT:%.*]], <16 x i32> [[A:%.*]], <16 x i32> [[B:%.*]], i32 [[SCALE_IDX:%.*]]) {
; GCN-NEXT: [[ENTRY:.*:]]
; GCN-NEXT: [[P0:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 0
-; GCN-NEXT: [[P4:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 4
-; GCN-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr addrspace(1) [[P0]], align 4
-; GCN-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr addrspace(1) [[P4]], align 4
-; GCN-NEXT: [[TMP2:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP0]])
-; GCN-NEXT: [[TMP3:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP1]])
-; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP2]])
-; GCN-NEXT: [[EXP1:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP3]])
+; GCN-NEXT: [[TMP0:%.*]] = load <8 x float>, ptr addrspace(1) [[P0]], align 4
+; GCN-NEXT: [[TMP1:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 0, i32 4>
+; GCN-NEXT: [[TMP2:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 1, i32 5>
+; GCN-NEXT: [[TMP3:%.*]] = fadd fast <2 x float> [[TMP1]], [[TMP2]]
+; GCN-NEXT: [[TMP4:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 2, i32 6>
+; GCN-NEXT: [[TMP5:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 3, i32 7>
+; GCN-NEXT: [[TMP6:%.*]] = fadd fast <2 x float> [[TMP4]], [[TMP5]]
+; GCN-NEXT: [[TMP7:%.*]] = fadd fast <2 x float> [[TMP3]], [[TMP6]]
+; GCN-NEXT: [[TMP8:%.*]] = extractelement <2 x float> [[TMP7]], i64 0
+; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP8]])
+; GCN-NEXT: [[TMP9:%.*]] = extractelement <2 x float> [[TMP7]], i64 1
+; GCN-NEXT: [[EXP1:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP9]])
; GCN-NEXT: [[VEC0:%.*]] = insertelement <2 x float> poison, float [[EXP0]], i64 0
; GCN-NEXT: [[VEC1:%.*]] = insertelement <2 x float> [[VEC0]], float [[EXP1]], i64 1
; GCN-NEXT: [[VEC_I32:%.*]] = bitcast <2 x float> [[VEC1]] to <2 x i32>
diff --git a/llvm/test/Transforms/VectorCombine/AMDGPU/combine-scalar-selects.ll b/llvm/test/Transforms/VectorCombine/AMDGPU/combine-scalar-selects.ll
index 78b58dc3e02fb..97ee47877ec2d 100644
--- a/llvm/test/Transforms/VectorCombine/AMDGPU/combine-scalar-selects.ll
+++ b/llvm/test/Transforms/VectorCombine/AMDGPU/combine-scalar-selects.ll
@@ -1024,31 +1024,38 @@ define amdgpu_kernel void @combine_v4f32_to_v8i16(
; CHECK-OPT-LABEL: define amdgpu_kernel void @combine_v4f32_to_v8i16(
; CHECK-OPT-SAME: ptr addrspace(1) [[OUT:%.*]], <4 x float> [[SRC:%.*]], i1 [[COND:%.*]]) {
; CHECK-OPT-NEXT: [[ENTRY:.*:]]
-; CHECK-OPT-NEXT: [[COMBINED_SEL:%.*]] = select i1 [[COND]], <4 x float> [[SRC]], <4 x float> zeroinitializer
-; CHECK-OPT-NEXT: [[COMBINED_BC:%.*]] = bitcast <4 x float> [[COMBINED_SEL]] to <8 x i16>
-; CHECK-OPT-NEXT: [[TMP0:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 0
-; CHECK-OPT-NEXT: [[TMP3:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 1
-; CHECK-OPT-NEXT: [[TMP5:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 2
-; CHECK-OPT-NEXT: [[TMP7:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 3
-; CHECK-OPT-NEXT: [[TMP2:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 4
-; CHECK-OPT-NEXT: [[TMP4:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 5
-; CHECK-OPT-NEXT: [[TMP6:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 6
-; CHECK-OPT-NEXT: [[TMP1:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 7
-; CHECK-OPT-NEXT: store i16 [[TMP0]], ptr addrspace(1) [[OUT]], align 2
+; CHECK-OPT-NEXT: [[HALVES:%.*]] = bitcast <4 x float> [[SRC]] to <8 x i16>
+; CHECK-OPT-NEXT: [[E0:%.*]] = extractelement <8 x i16> [[HALVES]], i64 0
+; CHECK-OPT-NEXT: [[E1:%.*]] = extractelement <8 x i16> [[HALVES]], i64 1
+; CHECK-OPT-NEXT: [[E2:%.*]] = extractelement <8 x i16> [[HALVES]], i64 2
+; CHECK-OPT-NEXT: [[E3:%.*]] = extractelement <8 x i16> [[HALVES]], i64 3
+; CHECK-OPT-NEXT: [[E4:%.*]] = extractelement <8 x i16> [[HALVES]], i64 4
+; CHECK-OPT-NEXT: [[E5:%.*]] = extractelement <8 x i16> [[HALVES]], i64 5
+; CHECK-OPT-NEXT: [[E6:%.*]] = extractelement <8 x i16> [[HALVES]], i64 6
+; CHECK-OPT-NEXT: [[E7:%.*]] = extractelement <8 x i16> [[HALVES]], i64 7
+; CHECK-OPT-NEXT: [[S0:%.*]] = select i1 [[COND]], i16 [[E0]], i16 0
+; CHECK-OPT-NEXT: [[S1:%.*]] = select i1 [[COND]], i16 [[E1]], i16 0
+; CHECK-OPT-NEXT: [[S2:%.*]] = select i1 [[COND]], i16 [[E2]], i16 0
+; CHECK-OPT-NEXT: [[S3:%.*]] = select i1 [[COND]], i16 [[E3]], i16 0
+; CHECK-OPT-NEXT: [[S4:%.*]] = select i1 [[COND]], i16 [[E4]], i16 0
+; CHECK-OPT-NEXT: [[S5:%.*]] = select i1 [[COND]], i16 [[E5]], i16 0
+; CHECK-OPT-NEXT: [[S6:%.*]] = select i1 [[COND]], i16 [[E6]], i16 0
+; CHECK-OPT-NEXT: [[S7:%.*]] = select i1 [[COND]], i16 [[E7]], i16 0
+; CHECK-OPT-NEXT: store i16 [[S0]], ptr addrspace(1) [[OUT]], align 2
; CHECK-OPT-NEXT: [[PTR1:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 1
-; CHECK-OPT-NEXT: store i16 [[TMP3]], ptr addrspace(1) [[PTR1]], align 2
+; CHECK-OPT-NEXT: store i16 [[S1]], ptr addrspace(1) [[PTR1]], align 2
; CHECK-OPT-NEXT: [[PTR2:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 2
-; CHECK-OPT-NEXT: store i16 [[TMP5]], ptr addrspace(1) [[PTR2]], align 2
+; CHECK-OPT-NEXT: store i16 [[S2]], ptr addrspace(1) [[PTR2]], align 2
; CHECK-OPT-NEXT: [[PTR3:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 3
-; CHECK-OPT-NEXT: store i16 [[TMP7]], ptr addrspace(1) [[PTR3]], align 2
+; CHECK-OPT-NEXT: store i16 [[S3]], ptr addrspace(1) [[PTR3]], align 2
; CHECK-OPT-NEXT: [[PTR4:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 4
-; CHECK-OPT-NEXT: store i16 [[TMP2]], ptr addrspace(1) [[PTR4]], align 2
+; CHECK-OPT-NEXT: store i16 [[S4]], ptr addrspace(1) [[PTR4]], align 2
; CHECK-OPT-NEXT: [[PTR5:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 5
-; CHECK-OPT-NEXT: store i16 [[TMP4]], ptr addrspace(1) [[PTR5]], align 2
+; CHECK-OPT-NEXT: store i16 [[S5]], ptr addrspace(1) [[PTR5]], align 2
; CHECK-OPT-NEXT: [[PTR6:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 6
-; CHECK-OPT-NEXT: store i16 [[TMP6]], ptr addrspace(1) [[PTR6]], align 2
+; CHECK-OPT-NEXT: store i16 [[S6]], ptr addrspace(1) [[PTR6]], align 2
; CHECK-OPT-NEXT: [[PTR7:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 7
-; CHECK-OPT-NEXT: store i16 [[TMP1]], ptr addrspace(1) [[PTR7]], align 2
+; CHECK-OPT-NEXT: store i16 [[S7]], ptr addrspace(1) [[PTR7]], align 2
; CHECK-OPT-NEXT: ret void
;
; CHECK-NOOPT-LABEL: define amdgpu_kernel void @combine_v4f32_to_v8i16(
>From 697378a3af6bf522d0cd9bc87b8524b64089b6fc Mon Sep 17 00:00:00 2001
From: Akash Dutta <Akash.Dutta at amd.com>
Date: Thu, 30 Jul 2026 19:25:53 +0000
Subject: [PATCH 06/11] remove GFX9 check on shuffle cost
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 8 ++---
...otriviallyvectorizableintrinsicoperands.ll | 33 ++++++-------------
2 files changed, 14 insertions(+), 27 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index b62854f1b78cf..5adef960f8e4a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1357,11 +1357,11 @@ InstructionCost GCNTTIImpl::getShuffleCost(TTI::ShuffleKind Kind,
unsigned ScalarSize = DL.getTypeSizeInBits(SrcTy->getElementType());
- // Gfx9 packed FP32 shuffles are free. InsertElement above already taxes
- // assembling <2 x f32> pairs, and a per-lane shuffle cost stacks on top and
- // over-penalizes SLP. f32-only on gfx9 targets with packed FP32 ops.
+ // Packed FP32 shuffles are free. InsertElement above already taxes assembling
+ // <2 x f32> pairs, and a per-lane shuffle cost stacks on top and over-penalizes
+ // SLP. f32-only on targets with packed FP32 ops.
if (ScalarSize == 32 && SrcTy->getElementType()->isFloatTy() &&
- ST->hasPackedFP32Ops() && ST->getGeneration() == AMDGPUSubtarget::GFX9) {
+ ST->hasPackedFP32Ops()) {
return 0;
}
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
index 4678c506846ba..533f8bce9ed18 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
@@ -542,17 +542,9 @@ define amdgpu_kernel void @test_single_exp_hreduction(
; GCN-SAME: ptr addrspace(1) [[INPUT:%.*]], ptr addrspace(1) [[OUTPUT:%.*]]) {
; GCN-NEXT: [[ENTRY:.*:]]
; GCN-NEXT: [[P0:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 0
-; GCN-NEXT: [[P1:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 1
-; GCN-NEXT: [[P2:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 2
-; GCN-NEXT: [[P3:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 3
-; GCN-NEXT: [[A0:%.*]] = load float, ptr addrspace(1) [[P0]], align 4
-; GCN-NEXT: [[A1:%.*]] = load float, ptr addrspace(1) [[P1]], align 4
-; GCN-NEXT: [[A2:%.*]] = load float, ptr addrspace(1) [[P2]], align 4
-; GCN-NEXT: [[A3:%.*]] = load float, ptr addrspace(1) [[P3]], align 4
-; GCN-NEXT: [[ADD01:%.*]] = fadd fast float [[A0]], [[A1]]
-; GCN-NEXT: [[ADD23:%.*]] = fadd fast float [[A2]], [[A3]]
-; GCN-NEXT: [[SUM:%.*]] = fadd fast float [[ADD01]], [[ADD23]]
-; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[SUM]])
+; GCN-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr addrspace(1) [[P0]], align 4
+; GCN-NEXT: [[TMP1:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP0]])
+; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP1]])
; GCN-NEXT: store float [[EXP0]], ptr addrspace(1) [[OUTPUT]], align 4
; GCN-NEXT: ret void
;
@@ -582,18 +574,13 @@ define amdgpu_kernel void @test_hreduction_into_exp(
; GCN-SAME: ptr addrspace(1) [[INPUT:%.*]], ptr addrspace(1) [[OUTPUT:%.*]], <16 x i32> [[A:%.*]], <16 x i32> [[B:%.*]], i32 [[SCALE_IDX:%.*]]) {
; GCN-NEXT: [[ENTRY:.*:]]
; GCN-NEXT: [[P0:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 0
-; GCN-NEXT: [[TMP0:%.*]] = load <8 x float>, ptr addrspace(1) [[P0]], align 4
-; GCN-NEXT: [[TMP1:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 0, i32 4>
-; GCN-NEXT: [[TMP2:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 1, i32 5>
-; GCN-NEXT: [[TMP3:%.*]] = fadd fast <2 x float> [[TMP1]], [[TMP2]]
-; GCN-NEXT: [[TMP4:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 2, i32 6>
-; GCN-NEXT: [[TMP5:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 3, i32 7>
-; GCN-NEXT: [[TMP6:%.*]] = fadd fast <2 x float> [[TMP4]], [[TMP5]]
-; GCN-NEXT: [[TMP7:%.*]] = fadd fast <2 x float> [[TMP3]], [[TMP6]]
-; GCN-NEXT: [[TMP8:%.*]] = extractelement <2 x float> [[TMP7]], i64 0
-; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP8]])
-; GCN-NEXT: [[TMP9:%.*]] = extractelement <2 x float> [[TMP7]], i64 1
-; GCN-NEXT: [[EXP1:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP9]])
+; GCN-NEXT: [[P4:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 4
+; GCN-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr addrspace(1) [[P0]], align 4
+; GCN-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr addrspace(1) [[P4]], align 4
+; GCN-NEXT: [[TMP2:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP0]])
+; GCN-NEXT: [[TMP3:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP1]])
+; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP2]])
+; GCN-NEXT: [[EXP1:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP3]])
; GCN-NEXT: [[VEC0:%.*]] = insertelement <2 x float> poison, float [[EXP0]], i64 0
; GCN-NEXT: [[VEC1:%.*]] = insertelement <2 x float> [[VEC0]], float [[EXP1]], i64 1
; GCN-NEXT: [[VEC_I32:%.*]] = bitcast <2 x float> [[VEC1]] to <2 x i32>
>From 0ae4a5d7c9cf77b8c6fb78592268d0c8e5a84624 Mon Sep 17 00:00:00 2001
From: Akash Dutta <Akash.Dutta at amd.com>
Date: Thu, 30 Jul 2026 21:36:06 +0000
Subject: [PATCH 07/11] code format
---
llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 5adef960f8e4a..3f37f2dbf18a4 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1358,8 +1358,8 @@ InstructionCost GCNTTIImpl::getShuffleCost(TTI::ShuffleKind Kind,
unsigned ScalarSize = DL.getTypeSizeInBits(SrcTy->getElementType());
// Packed FP32 shuffles are free. InsertElement above already taxes assembling
- // <2 x f32> pairs, and a per-lane shuffle cost stacks on top and over-penalizes
- // SLP. f32-only on targets with packed FP32 ops.
+ // <2 x f32> pairs, and a per-lane shuffle cost stacks on top and
+ // over-penalizes SLP. f32-only on targets with packed FP32 ops.
if (ScalarSize == 32 && SrcTy->getElementType()->isFloatTy() &&
ST->hasPackedFP32Ops()) {
return 0;
>From 39f7a7a8ff4672b6a0986bc04fc113660089f398 Mon Sep 17 00:00:00 2001
From: Akash Dutta <Akash.Dutta at amd.com>
Date: Thu, 30 Jul 2026 21:47:46 +0000
Subject: [PATCH 08/11] fix unconditional free shuffles
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 25 ++++--
.../CostModel/AMDGPU/packed-fp32-shuffle.ll | 89 +++++++++++++++++++
...otriviallyvectorizableintrinsicoperands.ll | 33 ++++---
3 files changed, 131 insertions(+), 16 deletions(-)
create mode 100644 llvm/test/Analysis/CostModel/AMDGPU/packed-fp32-shuffle.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 3f37f2dbf18a4..ee49e32c819c1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1357,12 +1357,25 @@ InstructionCost GCNTTIImpl::getShuffleCost(TTI::ShuffleKind Kind,
unsigned ScalarSize = DL.getTypeSizeInBits(SrcTy->getElementType());
- // Packed FP32 shuffles are free. InsertElement above already taxes assembling
- // <2 x f32> pairs, and a per-lane shuffle cost stacks on top and
- // over-penalizes SLP. f32-only on targets with packed FP32 ops.
- if (ScalarSize == 32 && SrcTy->getElementType()->isFloatTy() &&
- ST->hasPackedFP32Ops()) {
- return 0;
+ // Legal <2 x f32> shuffles for v_pk_*_f32 sources. Identity and broadcasts
+ // are free via op_sel on packed ops; high-to-low lane swap within an aligned
+ // pair is lowered to v_pk_mov_b32.
+ if (ST->hasPackedFP32Ops() && ScalarSize == 32) {
+ auto *DstVecTy = dyn_cast<FixedVectorType>(DstTy);
+ auto *SrcVecTy = dyn_cast<FixedVectorType>(SrcTy);
+ if (DstVecTy && SrcVecTy && DstVecTy->getNumElements() == 2 &&
+ SrcVecTy->getNumElements() == 2 &&
+ DstVecTy->getElementType()->isFloatTy()) {
+ switch (Kind) {
+ case TTI::SK_Broadcast:
+ case TTI::SK_PermuteSingleSrc:
+ return 0;
+ case TTI::SK_Reverse:
+ return 1;
+ default:
+ break;
+ }
+ }
}
if (ST->getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS &&
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/packed-fp32-shuffle.ll b/llvm/test/Analysis/CostModel/AMDGPU/packed-fp32-shuffle.ll
new file mode 100644
index 0000000000000..949275aff9798
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/AMDGPU/packed-fp32-shuffle.ll
@@ -0,0 +1,89 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.0a-unknown-amdhsa -S | FileCheck -check-prefixes=GFX90A %s
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -S | FileCheck -check-prefixes=GFX900 %s
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu8.03-unknown-amdhsa -S | FileCheck -check-prefixes=VI %s
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.0a-unknown-amdhsa -cost-kind=code-size -S | FileCheck -check-prefixes=GFX90A-SIZE %s
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -cost-kind=code-size -S | FileCheck -check-prefixes=GFX900-SIZE %s
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu8.03-unknown-amdhsa -cost-kind=code-size -S | FileCheck -check-prefixes=VI-SIZE %s
+; END.
+
+; Costs for legal <2 x f32> shuffles used to form v_pk_*_f32 sources. Identity and
+; broadcasts are free via op_sel; high-to-low lane swap needs v_pk_mov_b32.
+
+define amdgpu_kernel void @packed_fp32_shufflevector(<2 x float> %vec1, <2 x float> %vec2) {
+; GFX90A-LABEL: 'packed_fp32_shufflevector'
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; GFX900-LABEL: 'packed_fp32_shufflevector'
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+; GFX900-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; VI-LABEL: 'packed_fp32_shufflevector'
+; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
+; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
+; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
+; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
+; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
+; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
+; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
+; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+; VI-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; GFX90A-SIZE-LABEL: 'packed_fp32_shufflevector'
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; GFX900-SIZE-LABEL: 'packed_fp32_shufflevector'
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; VI-SIZE-LABEL: 'packed_fp32_shufflevector'
+; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
+; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
+; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
+; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
+; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
+; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
+; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
+; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
+ %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
+ %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
+ %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
+ %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
+ %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
+ %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
+ %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+ ret void
+}
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
index 533f8bce9ed18..4678c506846ba 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
@@ -542,9 +542,17 @@ define amdgpu_kernel void @test_single_exp_hreduction(
; GCN-SAME: ptr addrspace(1) [[INPUT:%.*]], ptr addrspace(1) [[OUTPUT:%.*]]) {
; GCN-NEXT: [[ENTRY:.*:]]
; GCN-NEXT: [[P0:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 0
-; GCN-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr addrspace(1) [[P0]], align 4
-; GCN-NEXT: [[TMP1:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP0]])
-; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP1]])
+; GCN-NEXT: [[P1:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 1
+; GCN-NEXT: [[P2:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 2
+; GCN-NEXT: [[P3:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 3
+; GCN-NEXT: [[A0:%.*]] = load float, ptr addrspace(1) [[P0]], align 4
+; GCN-NEXT: [[A1:%.*]] = load float, ptr addrspace(1) [[P1]], align 4
+; GCN-NEXT: [[A2:%.*]] = load float, ptr addrspace(1) [[P2]], align 4
+; GCN-NEXT: [[A3:%.*]] = load float, ptr addrspace(1) [[P3]], align 4
+; GCN-NEXT: [[ADD01:%.*]] = fadd fast float [[A0]], [[A1]]
+; GCN-NEXT: [[ADD23:%.*]] = fadd fast float [[A2]], [[A3]]
+; GCN-NEXT: [[SUM:%.*]] = fadd fast float [[ADD01]], [[ADD23]]
+; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[SUM]])
; GCN-NEXT: store float [[EXP0]], ptr addrspace(1) [[OUTPUT]], align 4
; GCN-NEXT: ret void
;
@@ -574,13 +582,18 @@ define amdgpu_kernel void @test_hreduction_into_exp(
; GCN-SAME: ptr addrspace(1) [[INPUT:%.*]], ptr addrspace(1) [[OUTPUT:%.*]], <16 x i32> [[A:%.*]], <16 x i32> [[B:%.*]], i32 [[SCALE_IDX:%.*]]) {
; GCN-NEXT: [[ENTRY:.*:]]
; GCN-NEXT: [[P0:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 0
-; GCN-NEXT: [[P4:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 4
-; GCN-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr addrspace(1) [[P0]], align 4
-; GCN-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr addrspace(1) [[P4]], align 4
-; GCN-NEXT: [[TMP2:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP0]])
-; GCN-NEXT: [[TMP3:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP1]])
-; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP2]])
-; GCN-NEXT: [[EXP1:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP3]])
+; GCN-NEXT: [[TMP0:%.*]] = load <8 x float>, ptr addrspace(1) [[P0]], align 4
+; GCN-NEXT: [[TMP1:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 0, i32 4>
+; GCN-NEXT: [[TMP2:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 1, i32 5>
+; GCN-NEXT: [[TMP3:%.*]] = fadd fast <2 x float> [[TMP1]], [[TMP2]]
+; GCN-NEXT: [[TMP4:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 2, i32 6>
+; GCN-NEXT: [[TMP5:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 3, i32 7>
+; GCN-NEXT: [[TMP6:%.*]] = fadd fast <2 x float> [[TMP4]], [[TMP5]]
+; GCN-NEXT: [[TMP7:%.*]] = fadd fast <2 x float> [[TMP3]], [[TMP6]]
+; GCN-NEXT: [[TMP8:%.*]] = extractelement <2 x float> [[TMP7]], i64 0
+; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP8]])
+; GCN-NEXT: [[TMP9:%.*]] = extractelement <2 x float> [[TMP7]], i64 1
+; GCN-NEXT: [[EXP1:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP9]])
; GCN-NEXT: [[VEC0:%.*]] = insertelement <2 x float> poison, float [[EXP0]], i64 0
; GCN-NEXT: [[VEC1:%.*]] = insertelement <2 x float> [[VEC0]], float [[EXP1]], i64 1
; GCN-NEXT: [[VEC_I32:%.*]] = bitcast <2 x float> [[VEC1]] to <2 x i32>
>From 11b56e5f7b3219e5a6581b420755b358c2503c77 Mon Sep 17 00:00:00 2001
From: Akash Dutta <Akash.Dutta at amd.com>
Date: Fri, 31 Jul 2026 13:58:09 +0000
Subject: [PATCH 09/11] update tests after main merge
---
.../irreducible/diverged-entry-basic-gmir.mir | 2 +-
.../irreducible/diverged-entry-basic.ll | 2 +-
...otriviallyvectorizableintrinsicoperands.ll | 27 +-
.../AMDGPU/ordered-reduction-fma-fusion.ll | 392 +++++++++++-------
4 files changed, 241 insertions(+), 182 deletions(-)
diff --git a/llvm/test/Analysis/UniformityAnalysis/AMDGPU/MIR/irreducible/diverged-entry-basic-gmir.mir b/llvm/test/Analysis/UniformityAnalysis/AMDGPU/MIR/irreducible/diverged-entry-basic-gmir.mir
index e3233b7ef6381..b6de0d6ec093d 100644
--- a/llvm/test/Analysis/UniformityAnalysis/AMDGPU/MIR/irreducible/diverged-entry-basic-gmir.mir
+++ b/llvm/test/Analysis/UniformityAnalysis/AMDGPU/MIR/irreducible/diverged-entry-basic-gmir.mir
@@ -4,8 +4,8 @@
# CHECK-NEXT: CYCLES ASSUMED DIVERGENT:
# CHECK-NEXT: depth=1: entries(bb.3 bb.1) bb.4 bb.2
# CHECK-NEXT: CYCLES WITH DIVERGENT EXIT:
-# CHECK-NEXT: depth=1: entries(bb.3 bb.1) bb.4 bb.2
# CHECK-NEXT: depth=2: entries(bb.4 bb.1) bb.2
+# CHECK-NEXT: depth=1: entries(bb.3 bb.1) bb.4 bb.2
diff --git a/llvm/test/Analysis/UniformityAnalysis/AMDGPU/irreducible/diverged-entry-basic.ll b/llvm/test/Analysis/UniformityAnalysis/AMDGPU/irreducible/diverged-entry-basic.ll
index 4f253ca0378f1..033ecac072cbe 100644
--- a/llvm/test/Analysis/UniformityAnalysis/AMDGPU/irreducible/diverged-entry-basic.ll
+++ b/llvm/test/Analysis/UniformityAnalysis/AMDGPU/irreducible/diverged-entry-basic.ll
@@ -5,8 +5,8 @@ define amdgpu_kernel void @divergent_cycle_1(i32 %a, i32 %b, i32 %c) {
; CHECK: CYCLES ASSUMED DIVERGENT:
; CHECK: depth=1: entries(R P) S Q
; CHECK: CYCLES WITH DIVERGENT EXIT:
-; CHECK: depth=1: entries(R P) S Q
; CHECK: depth=2: entries(S P) Q
+; CHECK: depth=1: entries(R P) S Q
entry:
%cond.uni = icmp slt i32 %a, 0
%tid = call i32 @llvm.amdgcn.workitem.id.x()
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
index 4678c506846ba..b2b5ac09a2426 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
@@ -542,16 +542,8 @@ define amdgpu_kernel void @test_single_exp_hreduction(
; GCN-SAME: ptr addrspace(1) [[INPUT:%.*]], ptr addrspace(1) [[OUTPUT:%.*]]) {
; GCN-NEXT: [[ENTRY:.*:]]
; GCN-NEXT: [[P0:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 0
-; GCN-NEXT: [[P1:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 1
-; GCN-NEXT: [[P2:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 2
-; GCN-NEXT: [[P3:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 3
-; GCN-NEXT: [[A0:%.*]] = load float, ptr addrspace(1) [[P0]], align 4
-; GCN-NEXT: [[A1:%.*]] = load float, ptr addrspace(1) [[P1]], align 4
-; GCN-NEXT: [[A2:%.*]] = load float, ptr addrspace(1) [[P2]], align 4
-; GCN-NEXT: [[A3:%.*]] = load float, ptr addrspace(1) [[P3]], align 4
-; GCN-NEXT: [[ADD01:%.*]] = fadd fast float [[A0]], [[A1]]
-; GCN-NEXT: [[ADD23:%.*]] = fadd fast float [[A2]], [[A3]]
-; GCN-NEXT: [[SUM:%.*]] = fadd fast float [[ADD01]], [[ADD23]]
+; GCN-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr addrspace(1) [[P0]], align 4
+; GCN-NEXT: [[SUM:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP0]])
; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[SUM]])
; GCN-NEXT: store float [[EXP0]], ptr addrspace(1) [[OUTPUT]], align 4
; GCN-NEXT: ret void
@@ -582,17 +574,12 @@ define amdgpu_kernel void @test_hreduction_into_exp(
; GCN-SAME: ptr addrspace(1) [[INPUT:%.*]], ptr addrspace(1) [[OUTPUT:%.*]], <16 x i32> [[A:%.*]], <16 x i32> [[B:%.*]], i32 [[SCALE_IDX:%.*]]) {
; GCN-NEXT: [[ENTRY:.*:]]
; GCN-NEXT: [[P0:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 0
-; GCN-NEXT: [[TMP0:%.*]] = load <8 x float>, ptr addrspace(1) [[P0]], align 4
-; GCN-NEXT: [[TMP1:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 0, i32 4>
-; GCN-NEXT: [[TMP2:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 1, i32 5>
-; GCN-NEXT: [[TMP3:%.*]] = fadd fast <2 x float> [[TMP1]], [[TMP2]]
-; GCN-NEXT: [[TMP4:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 2, i32 6>
-; GCN-NEXT: [[TMP5:%.*]] = shufflevector <8 x float> [[TMP0]], <8 x float> poison, <2 x i32> <i32 3, i32 7>
-; GCN-NEXT: [[TMP6:%.*]] = fadd fast <2 x float> [[TMP4]], [[TMP5]]
-; GCN-NEXT: [[TMP7:%.*]] = fadd fast <2 x float> [[TMP3]], [[TMP6]]
-; GCN-NEXT: [[TMP8:%.*]] = extractelement <2 x float> [[TMP7]], i64 0
+; GCN-NEXT: [[P4:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 4
+; GCN-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr addrspace(1) [[P0]], align 4
+; GCN-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr addrspace(1) [[P4]], align 4
+; GCN-NEXT: [[TMP8:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP0]])
+; GCN-NEXT: [[TMP9:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP1]])
; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP8]])
-; GCN-NEXT: [[TMP9:%.*]] = extractelement <2 x float> [[TMP7]], i64 1
; GCN-NEXT: [[EXP1:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP9]])
; GCN-NEXT: [[VEC0:%.*]] = insertelement <2 x float> poison, float [[EXP0]], i64 0
; GCN-NEXT: [[VEC1:%.*]] = insertelement <2 x float> [[VEC0]], float [[EXP1]], i64 1
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/ordered-reduction-fma-fusion.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/ordered-reduction-fma-fusion.ll
index 6a7a93a44f8c9..957c7a532d590 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/ordered-reduction-fma-fusion.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/ordered-reduction-fma-fusion.ll
@@ -9,103 +9,139 @@
define float @conv_contract(ptr addrspace(1) %input, ptr addrspace(4) %mask) {
; CHECK-LABEL: @conv_contract(
; CHECK-NEXT: [[IP0:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT:%.*]], i64 0
+; CHECK-NEXT: [[IV0:%.*]] = load float, ptr addrspace(1) [[IP0]], align 4
; CHECK-NEXT: [[MP0:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK:%.*]], i64 0
+; CHECK-NEXT: [[MV0:%.*]] = load float, ptr addrspace(4) [[MP0]], align 4
+; CHECK-NEXT: [[PROD0:%.*]] = fmul contract float [[IV0]], [[MV0]]
+; CHECK-NEXT: [[IP1:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 1
+; CHECK-NEXT: [[IV1:%.*]] = load float, ptr addrspace(1) [[IP1]], align 4
+; CHECK-NEXT: [[MP1:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 1
+; CHECK-NEXT: [[MV1:%.*]] = load float, ptr addrspace(4) [[MP1]], align 4
+; CHECK-NEXT: [[PROD1:%.*]] = fmul contract float [[IV1]], [[MV1]]
+; CHECK-NEXT: [[ACC1:%.*]] = fadd contract float [[PROD0]], [[PROD1]]
+; CHECK-NEXT: [[IP2:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 2
+; CHECK-NEXT: [[IV2:%.*]] = load float, ptr addrspace(1) [[IP2]], align 4
+; CHECK-NEXT: [[MP2:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 2
+; CHECK-NEXT: [[MV2:%.*]] = load float, ptr addrspace(4) [[MP2]], align 4
+; CHECK-NEXT: [[PROD2:%.*]] = fmul contract float [[IV2]], [[MV2]]
+; CHECK-NEXT: [[ACC2:%.*]] = fadd contract float [[ACC1]], [[PROD2]]
+; CHECK-NEXT: [[IP3:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 3
+; CHECK-NEXT: [[IV3:%.*]] = load float, ptr addrspace(1) [[IP3]], align 4
+; CHECK-NEXT: [[MP3:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 3
+; CHECK-NEXT: [[MV3:%.*]] = load float, ptr addrspace(4) [[MP3]], align 4
+; CHECK-NEXT: [[PROD3:%.*]] = fmul contract float [[IV3]], [[MV3]]
+; CHECK-NEXT: [[ACC3:%.*]] = fadd contract float [[ACC2]], [[PROD3]]
; CHECK-NEXT: [[IP4:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 4
; CHECK-NEXT: [[IV4:%.*]] = load float, ptr addrspace(1) [[IP4]], align 4
+; CHECK-NEXT: [[MP4:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 4
+; CHECK-NEXT: [[MV4:%.*]] = load float, ptr addrspace(4) [[MP4]], align 4
+; CHECK-NEXT: [[PROD4:%.*]] = fmul contract float [[IV4]], [[MV4]]
+; CHECK-NEXT: [[ACC4:%.*]] = fadd contract float [[ACC3]], [[PROD4]]
; CHECK-NEXT: [[IP5:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 8
; CHECK-NEXT: [[IV5:%.*]] = load float, ptr addrspace(1) [[IP5]], align 4
+; CHECK-NEXT: [[MP5:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 5
+; CHECK-NEXT: [[MV5:%.*]] = load float, ptr addrspace(4) [[MP5]], align 4
+; CHECK-NEXT: [[PROD5:%.*]] = fmul contract float [[IV5]], [[MV5]]
+; CHECK-NEXT: [[ACC5:%.*]] = fadd contract float [[ACC4]], [[PROD5]]
; CHECK-NEXT: [[IP6:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 9
+; CHECK-NEXT: [[IV6:%.*]] = load float, ptr addrspace(1) [[IP6]], align 4
+; CHECK-NEXT: [[MP6:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 6
+; CHECK-NEXT: [[MV6:%.*]] = load float, ptr addrspace(4) [[MP6]], align 4
+; CHECK-NEXT: [[PROD6:%.*]] = fmul contract float [[IV6]], [[MV6]]
+; CHECK-NEXT: [[ACC6:%.*]] = fadd contract float [[ACC5]], [[PROD6]]
+; CHECK-NEXT: [[IP7:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 10
+; CHECK-NEXT: [[IV7:%.*]] = load float, ptr addrspace(1) [[IP7]], align 4
+; CHECK-NEXT: [[MP7:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 7
+; CHECK-NEXT: [[MV7:%.*]] = load float, ptr addrspace(4) [[MP7]], align 4
+; CHECK-NEXT: [[PROD7:%.*]] = fmul contract float [[IV7]], [[MV7]]
+; CHECK-NEXT: [[ACC7:%.*]] = fadd contract float [[ACC6]], [[PROD7]]
; CHECK-NEXT: [[IP8:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 11
+; CHECK-NEXT: [[IV8:%.*]] = load float, ptr addrspace(1) [[IP8]], align 4
+; CHECK-NEXT: [[MP8:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 8
+; CHECK-NEXT: [[MV8:%.*]] = load float, ptr addrspace(4) [[MP8]], align 4
+; CHECK-NEXT: [[PROD8:%.*]] = fmul contract float [[IV8]], [[MV8]]
+; CHECK-NEXT: [[ACC8:%.*]] = fadd contract float [[ACC7]], [[PROD8]]
+; CHECK-NEXT: [[IP9:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 12
+; CHECK-NEXT: [[IV9:%.*]] = load float, ptr addrspace(1) [[IP9]], align 4
+; CHECK-NEXT: [[MP9:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 9
+; CHECK-NEXT: [[MV9:%.*]] = load float, ptr addrspace(4) [[MP9]], align 4
+; CHECK-NEXT: [[PROD9:%.*]] = fmul contract float [[IV9]], [[MV9]]
+; CHECK-NEXT: [[ACC9:%.*]] = fadd contract float [[ACC8]], [[PROD9]]
; CHECK-NEXT: [[IP10:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 16
+; CHECK-NEXT: [[IV10:%.*]] = load float, ptr addrspace(1) [[IP10]], align 4
+; CHECK-NEXT: [[MP10:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 10
+; CHECK-NEXT: [[MV10:%.*]] = load float, ptr addrspace(4) [[MP10]], align 4
+; CHECK-NEXT: [[PROD10:%.*]] = fmul contract float [[IV10]], [[MV10]]
+; CHECK-NEXT: [[ACC10:%.*]] = fadd contract float [[ACC9]], [[PROD10]]
+; CHECK-NEXT: [[IP11:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 17
+; CHECK-NEXT: [[IV11:%.*]] = load float, ptr addrspace(1) [[IP11]], align 4
+; CHECK-NEXT: [[MP11:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 11
+; CHECK-NEXT: [[MV11:%.*]] = load float, ptr addrspace(4) [[MP11]], align 4
+; CHECK-NEXT: [[PROD11:%.*]] = fmul contract float [[IV11]], [[MV11]]
+; CHECK-NEXT: [[ACC11:%.*]] = fadd contract float [[ACC10]], [[PROD11]]
; CHECK-NEXT: [[IP12:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 18
-; CHECK-NEXT: [[IP14:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 20
-; CHECK-NEXT: [[IV14:%.*]] = load float, ptr addrspace(1) [[IP14]], align 4
+; CHECK-NEXT: [[IV12:%.*]] = load float, ptr addrspace(1) [[IP12]], align 4
+; CHECK-NEXT: [[MP12:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 12
+; CHECK-NEXT: [[MV12:%.*]] = load float, ptr addrspace(4) [[MP12]], align 4
+; CHECK-NEXT: [[PROD12:%.*]] = fmul contract float [[IV12]], [[MV12]]
+; CHECK-NEXT: [[ACC12:%.*]] = fadd contract float [[ACC11]], [[PROD12]]
+; CHECK-NEXT: [[IP13:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 19
+; CHECK-NEXT: [[MP13:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 13
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x float>, ptr addrspace(1) [[IP13]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = load <2 x float>, ptr addrspace(4) [[MP13]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = fmul contract <2 x float> [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x float> [[TMP3]], i64 0
+; CHECK-NEXT: [[ACC13:%.*]] = fadd contract float [[ACC12]], [[TMP4]]
+; CHECK-NEXT: [[TMP5:%.*]] = extractelement <2 x float> [[TMP3]], i64 1
+; CHECK-NEXT: [[ACC25:%.*]] = fadd contract float [[ACC13]], [[TMP5]]
; CHECK-NEXT: [[IP15:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 24
; CHECK-NEXT: [[IV15:%.*]] = load float, ptr addrspace(1) [[IP15]], align 4
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr addrspace(1) [[IP0]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = load <2 x float>, ptr addrspace(1) [[IP6]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = load <2 x float>, ptr addrspace(1) [[IP8]], align 4
-; CHECK-NEXT: [[TMP4:%.*]] = load <2 x float>, ptr addrspace(1) [[IP10]], align 4
-; CHECK-NEXT: [[TMP5:%.*]] = load <2 x float>, ptr addrspace(1) [[IP12]], align 4
-; CHECK-NEXT: [[TMP6:%.*]] = load <16 x float>, ptr addrspace(4) [[MP0]], align 4
-; CHECK-NEXT: [[TMP7:%.*]] = insertelement <16 x float> poison, float [[IV4]], i64 4
-; CHECK-NEXT: [[TMP8:%.*]] = insertelement <16 x float> [[TMP7]], float [[IV5]], i64 5
-; CHECK-NEXT: [[TMP22:%.*]] = insertelement <16 x float> [[TMP8]], float [[IV14]], i64 14
-; CHECK-NEXT: [[TMP23:%.*]] = insertelement <16 x float> [[TMP22]], float [[IV15]], i64 15
-; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <16 x float> [[TMP23]], <16 x float> [[TMP11]], <16 x i32> <i32 16, i32 17, i32 18, i32 19, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
-; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <2 x float> [[TMP2]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP24:%.*]] = shufflevector <16 x float> [[TMP12]], <16 x float> [[TMP13]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 16, i32 17, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
-; CHECK-NEXT: [[TMP25:%.*]] = shufflevector <2 x float> [[TMP3]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP16:%.*]] = shufflevector <16 x float> [[TMP24]], <16 x float> [[TMP25]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 16, i32 17, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
-; CHECK-NEXT: [[TMP17:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP18:%.*]] = shufflevector <16 x float> [[TMP16]], <16 x float> [[TMP17]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 16, i32 17, i32 12, i32 13, i32 14, i32 15>
-; CHECK-NEXT: [[TMP19:%.*]] = shufflevector <2 x float> [[TMP5]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP20:%.*]] = shufflevector <16 x float> [[TMP18]], <16 x float> [[TMP19]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 16, i32 17, i32 14, i32 15>
-; CHECK-NEXT: [[TMP21:%.*]] = fmul contract <16 x float> [[TMP20]], [[TMP6]]
-; CHECK-NEXT: [[ACC14:%.*]] = extractelement <16 x float> [[TMP21]], i64 0
-; CHECK-NEXT: [[PROD15:%.*]] = extractelement <16 x float> [[TMP21]], i64 1
-; CHECK-NEXT: [[ACC15:%.*]] = fadd contract float [[ACC14]], [[PROD15]]
-; CHECK-NEXT: [[TMP9:%.*]] = extractelement <16 x float> [[TMP21]], i64 2
-; CHECK-NEXT: [[ACC16:%.*]] = fadd contract float [[ACC15]], [[TMP9]]
-; CHECK-NEXT: [[TMP10:%.*]] = extractelement <16 x float> [[TMP21]], i64 3
-; CHECK-NEXT: [[ACC17:%.*]] = fadd contract float [[ACC16]], [[TMP10]]
-; CHECK-NEXT: [[TMP14:%.*]] = extractelement <16 x float> [[TMP21]], i64 4
-; CHECK-NEXT: [[ACC18:%.*]] = fadd contract float [[ACC17]], [[TMP14]]
-; CHECK-NEXT: [[TMP15:%.*]] = extractelement <16 x float> [[TMP21]], i64 5
-; CHECK-NEXT: [[ACC19:%.*]] = fadd contract float [[ACC18]], [[TMP15]]
-; CHECK-NEXT: [[TMP28:%.*]] = extractelement <16 x float> [[TMP21]], i64 6
-; CHECK-NEXT: [[ACC6:%.*]] = fadd contract float [[ACC19]], [[TMP28]]
-; CHECK-NEXT: [[TMP29:%.*]] = extractelement <16 x float> [[TMP21]], i64 7
-; CHECK-NEXT: [[ACC7:%.*]] = fadd contract float [[ACC6]], [[TMP29]]
-; CHECK-NEXT: [[TMP30:%.*]] = extractelement <16 x float> [[TMP21]], i64 8
-; CHECK-NEXT: [[ACC8:%.*]] = fadd contract float [[ACC7]], [[TMP30]]
-; CHECK-NEXT: [[TMP31:%.*]] = extractelement <16 x float> [[TMP21]], i64 9
-; CHECK-NEXT: [[ACC9:%.*]] = fadd contract float [[ACC8]], [[TMP31]]
-; CHECK-NEXT: [[TMP32:%.*]] = extractelement <16 x float> [[TMP21]], i64 10
-; CHECK-NEXT: [[ACC10:%.*]] = fadd contract float [[ACC9]], [[TMP32]]
-; CHECK-NEXT: [[TMP33:%.*]] = extractelement <16 x float> [[TMP21]], i64 11
-; CHECK-NEXT: [[ACC11:%.*]] = fadd contract float [[ACC10]], [[TMP33]]
-; CHECK-NEXT: [[TMP34:%.*]] = extractelement <16 x float> [[TMP21]], i64 12
-; CHECK-NEXT: [[ACC12:%.*]] = fadd contract float [[ACC11]], [[TMP34]]
-; CHECK-NEXT: [[TMP35:%.*]] = extractelement <16 x float> [[TMP21]], i64 13
-; CHECK-NEXT: [[ACC13:%.*]] = fadd contract float [[ACC12]], [[TMP35]]
-; CHECK-NEXT: [[TMP36:%.*]] = extractelement <16 x float> [[TMP21]], i64 14
-; CHECK-NEXT: [[ACC25:%.*]] = fadd contract float [[ACC13]], [[TMP36]]
-; CHECK-NEXT: [[TMP37:%.*]] = extractelement <16 x float> [[TMP21]], i64 15
+; CHECK-NEXT: [[MP15:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 15
+; CHECK-NEXT: [[MV15:%.*]] = load float, ptr addrspace(4) [[MP15]], align 4
+; CHECK-NEXT: [[TMP37:%.*]] = fmul contract float [[IV15]], [[MV15]]
; CHECK-NEXT: [[ACC26:%.*]] = fadd contract float [[ACC25]], [[TMP37]]
; CHECK-NEXT: [[IP16:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 25
; CHECK-NEXT: [[MP16:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 16
+; CHECK-NEXT: [[TMP6:%.*]] = load <2 x float>, ptr addrspace(1) [[IP16]], align 4
+; CHECK-NEXT: [[TMP7:%.*]] = load <2 x float>, ptr addrspace(4) [[MP16]], align 4
+; CHECK-NEXT: [[TMP8:%.*]] = fmul contract <2 x float> [[TMP6]], [[TMP7]]
+; CHECK-NEXT: [[TMP9:%.*]] = extractelement <2 x float> [[TMP8]], i64 0
+; CHECK-NEXT: [[ACC16:%.*]] = fadd contract float [[ACC26]], [[TMP9]]
+; CHECK-NEXT: [[TMP10:%.*]] = extractelement <2 x float> [[TMP8]], i64 1
+; CHECK-NEXT: [[ACC17:%.*]] = fadd contract float [[ACC16]], [[TMP10]]
+; CHECK-NEXT: [[IP18:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 27
+; CHECK-NEXT: [[MP18:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 18
+; CHECK-NEXT: [[TMP11:%.*]] = load <2 x float>, ptr addrspace(1) [[IP18]], align 4
+; CHECK-NEXT: [[TMP12:%.*]] = load <2 x float>, ptr addrspace(4) [[MP18]], align 4
+; CHECK-NEXT: [[TMP13:%.*]] = fmul contract <2 x float> [[TMP11]], [[TMP12]]
+; CHECK-NEXT: [[TMP14:%.*]] = extractelement <2 x float> [[TMP13]], i64 0
+; CHECK-NEXT: [[ACC18:%.*]] = fadd contract float [[ACC17]], [[TMP14]]
+; CHECK-NEXT: [[TMP15:%.*]] = extractelement <2 x float> [[TMP13]], i64 1
+; CHECK-NEXT: [[ACC19:%.*]] = fadd contract float [[ACC18]], [[TMP15]]
; CHECK-NEXT: [[IP20:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 32
-; CHECK-NEXT: [[TMP38:%.*]] = load <4 x float>, ptr addrspace(1) [[IP16]], align 4
-; CHECK-NEXT: [[TMP39:%.*]] = load <4 x float>, ptr addrspace(1) [[IP20]], align 4
-; CHECK-NEXT: [[TMP40:%.*]] = load <8 x float>, ptr addrspace(4) [[MP16]], align 4
-; CHECK-NEXT: [[TMP41:%.*]] = shufflevector <4 x float> [[TMP38]], <4 x float> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP42:%.*]] = shufflevector <4 x float> [[TMP39]], <4 x float> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP43:%.*]] = shufflevector <4 x float> [[TMP38]], <4 x float> [[TMP39]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; CHECK-NEXT: [[TMP44:%.*]] = fmul contract <8 x float> [[TMP43]], [[TMP40]]
-; CHECK-NEXT: [[TMP45:%.*]] = extractelement <8 x float> [[TMP44]], i64 0
-; CHECK-NEXT: [[ACC27:%.*]] = fadd contract float [[ACC26]], [[TMP45]]
-; CHECK-NEXT: [[TMP46:%.*]] = extractelement <8 x float> [[TMP44]], i64 1
-; CHECK-NEXT: [[ACC28:%.*]] = fadd contract float [[ACC27]], [[TMP46]]
-; CHECK-NEXT: [[TMP47:%.*]] = extractelement <8 x float> [[TMP44]], i64 2
-; CHECK-NEXT: [[ACC29:%.*]] = fadd contract float [[ACC28]], [[TMP47]]
-; CHECK-NEXT: [[TMP48:%.*]] = extractelement <8 x float> [[TMP44]], i64 3
-; CHECK-NEXT: [[ACC30:%.*]] = fadd contract float [[ACC29]], [[TMP48]]
-; CHECK-NEXT: [[TMP49:%.*]] = extractelement <8 x float> [[TMP44]], i64 4
-; CHECK-NEXT: [[ACC20:%.*]] = fadd contract float [[ACC30]], [[TMP49]]
-; CHECK-NEXT: [[TMP50:%.*]] = extractelement <8 x float> [[TMP44]], i64 5
-; CHECK-NEXT: [[ACC21:%.*]] = fadd contract float [[ACC20]], [[TMP50]]
-; CHECK-NEXT: [[TMP51:%.*]] = extractelement <8 x float> [[TMP44]], i64 6
-; CHECK-NEXT: [[ACC22:%.*]] = fadd contract float [[ACC21]], [[TMP51]]
-; CHECK-NEXT: [[TMP52:%.*]] = extractelement <8 x float> [[TMP44]], i64 7
-; CHECK-NEXT: [[ACC23:%.*]] = fadd contract float [[ACC22]], [[TMP52]]
-; CHECK-NEXT: [[IP24:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 36
-; CHECK-NEXT: [[IV20:%.*]] = load float, ptr addrspace(1) [[IP24]], align 4
-; CHECK-NEXT: [[MP20:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 24
+; CHECK-NEXT: [[IV20:%.*]] = load float, ptr addrspace(1) [[IP20]], align 4
+; CHECK-NEXT: [[MP20:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 20
; CHECK-NEXT: [[MV20:%.*]] = load float, ptr addrspace(4) [[MP20]], align 4
; CHECK-NEXT: [[PROD20:%.*]] = fmul contract float [[IV20]], [[MV20]]
-; CHECK-NEXT: [[ACC24:%.*]] = fadd contract float [[ACC23]], [[PROD20]]
+; CHECK-NEXT: [[ACC20:%.*]] = fadd contract float [[ACC19]], [[PROD20]]
+; CHECK-NEXT: [[IP21:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 33
+; CHECK-NEXT: [[MP21:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 21
+; CHECK-NEXT: [[TMP16:%.*]] = load <2 x float>, ptr addrspace(1) [[IP21]], align 4
+; CHECK-NEXT: [[TMP17:%.*]] = load <2 x float>, ptr addrspace(4) [[MP21]], align 4
+; CHECK-NEXT: [[TMP18:%.*]] = fmul contract <2 x float> [[TMP16]], [[TMP17]]
+; CHECK-NEXT: [[TMP19:%.*]] = extractelement <2 x float> [[TMP18]], i64 0
+; CHECK-NEXT: [[ACC21:%.*]] = fadd contract float [[ACC20]], [[TMP19]]
+; CHECK-NEXT: [[TMP20:%.*]] = extractelement <2 x float> [[TMP18]], i64 1
+; CHECK-NEXT: [[ACC22:%.*]] = fadd contract float [[ACC21]], [[TMP20]]
+; CHECK-NEXT: [[IP23:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 35
+; CHECK-NEXT: [[MP23:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 23
+; CHECK-NEXT: [[TMP21:%.*]] = load <2 x float>, ptr addrspace(1) [[IP23]], align 4
+; CHECK-NEXT: [[TMP22:%.*]] = load <2 x float>, ptr addrspace(4) [[MP23]], align 4
+; CHECK-NEXT: [[TMP23:%.*]] = fmul contract <2 x float> [[TMP21]], [[TMP22]]
+; CHECK-NEXT: [[TMP24:%.*]] = extractelement <2 x float> [[TMP23]], i64 0
+; CHECK-NEXT: [[ACC23:%.*]] = fadd contract float [[ACC22]], [[TMP24]]
+; CHECK-NEXT: [[TMP25:%.*]] = extractelement <2 x float> [[TMP23]], i64 1
+; CHECK-NEXT: [[ACC24:%.*]] = fadd contract float [[ACC23]], [[TMP25]]
; CHECK-NEXT: ret float [[ACC24]]
;
%ip0 = getelementptr inbounds float, ptr addrspace(1) %input, i64 0
@@ -263,103 +299,139 @@ define float @conv_contract(ptr addrspace(1) %input, ptr addrspace(4) %mask) {
define float @conv_no_contract(ptr addrspace(1) %input, ptr addrspace(4) %mask) {
; CHECK-LABEL: @conv_no_contract(
; CHECK-NEXT: [[IP0:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT:%.*]], i64 0
+; CHECK-NEXT: [[IV0:%.*]] = load float, ptr addrspace(1) [[IP0]], align 4
; CHECK-NEXT: [[MP0:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK:%.*]], i64 0
+; CHECK-NEXT: [[MV0:%.*]] = load float, ptr addrspace(4) [[MP0]], align 4
+; CHECK-NEXT: [[PROD0:%.*]] = fmul float [[IV0]], [[MV0]]
+; CHECK-NEXT: [[IP1:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 1
+; CHECK-NEXT: [[IV1:%.*]] = load float, ptr addrspace(1) [[IP1]], align 4
+; CHECK-NEXT: [[MP1:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 1
+; CHECK-NEXT: [[MV1:%.*]] = load float, ptr addrspace(4) [[MP1]], align 4
+; CHECK-NEXT: [[PROD1:%.*]] = fmul float [[IV1]], [[MV1]]
+; CHECK-NEXT: [[ACC1:%.*]] = fadd float [[PROD0]], [[PROD1]]
+; CHECK-NEXT: [[IP2:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 2
+; CHECK-NEXT: [[IV2:%.*]] = load float, ptr addrspace(1) [[IP2]], align 4
+; CHECK-NEXT: [[MP2:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 2
+; CHECK-NEXT: [[MV2:%.*]] = load float, ptr addrspace(4) [[MP2]], align 4
+; CHECK-NEXT: [[PROD2:%.*]] = fmul float [[IV2]], [[MV2]]
+; CHECK-NEXT: [[ACC2:%.*]] = fadd float [[ACC1]], [[PROD2]]
+; CHECK-NEXT: [[IP3:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 3
+; CHECK-NEXT: [[IV3:%.*]] = load float, ptr addrspace(1) [[IP3]], align 4
+; CHECK-NEXT: [[MP3:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 3
+; CHECK-NEXT: [[MV3:%.*]] = load float, ptr addrspace(4) [[MP3]], align 4
+; CHECK-NEXT: [[PROD3:%.*]] = fmul float [[IV3]], [[MV3]]
+; CHECK-NEXT: [[ACC3:%.*]] = fadd float [[ACC2]], [[PROD3]]
; CHECK-NEXT: [[IP4:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 4
; CHECK-NEXT: [[IV4:%.*]] = load float, ptr addrspace(1) [[IP4]], align 4
+; CHECK-NEXT: [[MP4:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 4
+; CHECK-NEXT: [[MV4:%.*]] = load float, ptr addrspace(4) [[MP4]], align 4
+; CHECK-NEXT: [[PROD4:%.*]] = fmul float [[IV4]], [[MV4]]
+; CHECK-NEXT: [[ACC4:%.*]] = fadd float [[ACC3]], [[PROD4]]
; CHECK-NEXT: [[IP5:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 8
; CHECK-NEXT: [[IV5:%.*]] = load float, ptr addrspace(1) [[IP5]], align 4
+; CHECK-NEXT: [[MP5:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 5
+; CHECK-NEXT: [[MV5:%.*]] = load float, ptr addrspace(4) [[MP5]], align 4
+; CHECK-NEXT: [[PROD5:%.*]] = fmul float [[IV5]], [[MV5]]
+; CHECK-NEXT: [[ACC5:%.*]] = fadd float [[ACC4]], [[PROD5]]
; CHECK-NEXT: [[IP6:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 9
+; CHECK-NEXT: [[IV6:%.*]] = load float, ptr addrspace(1) [[IP6]], align 4
+; CHECK-NEXT: [[MP6:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 6
+; CHECK-NEXT: [[MV6:%.*]] = load float, ptr addrspace(4) [[MP6]], align 4
+; CHECK-NEXT: [[PROD6:%.*]] = fmul float [[IV6]], [[MV6]]
+; CHECK-NEXT: [[ACC6:%.*]] = fadd float [[ACC5]], [[PROD6]]
+; CHECK-NEXT: [[IP7:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 10
+; CHECK-NEXT: [[IV7:%.*]] = load float, ptr addrspace(1) [[IP7]], align 4
+; CHECK-NEXT: [[MP7:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 7
+; CHECK-NEXT: [[MV7:%.*]] = load float, ptr addrspace(4) [[MP7]], align 4
+; CHECK-NEXT: [[PROD7:%.*]] = fmul float [[IV7]], [[MV7]]
+; CHECK-NEXT: [[ACC7:%.*]] = fadd float [[ACC6]], [[PROD7]]
; CHECK-NEXT: [[IP8:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 11
+; CHECK-NEXT: [[IV8:%.*]] = load float, ptr addrspace(1) [[IP8]], align 4
+; CHECK-NEXT: [[MP8:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 8
+; CHECK-NEXT: [[MV8:%.*]] = load float, ptr addrspace(4) [[MP8]], align 4
+; CHECK-NEXT: [[PROD8:%.*]] = fmul float [[IV8]], [[MV8]]
+; CHECK-NEXT: [[ACC8:%.*]] = fadd float [[ACC7]], [[PROD8]]
+; CHECK-NEXT: [[IP9:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 12
+; CHECK-NEXT: [[IV9:%.*]] = load float, ptr addrspace(1) [[IP9]], align 4
+; CHECK-NEXT: [[MP9:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 9
+; CHECK-NEXT: [[MV9:%.*]] = load float, ptr addrspace(4) [[MP9]], align 4
+; CHECK-NEXT: [[PROD9:%.*]] = fmul float [[IV9]], [[MV9]]
+; CHECK-NEXT: [[ACC9:%.*]] = fadd float [[ACC8]], [[PROD9]]
; CHECK-NEXT: [[IP10:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 16
+; CHECK-NEXT: [[IV10:%.*]] = load float, ptr addrspace(1) [[IP10]], align 4
+; CHECK-NEXT: [[MP10:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 10
+; CHECK-NEXT: [[MV10:%.*]] = load float, ptr addrspace(4) [[MP10]], align 4
+; CHECK-NEXT: [[PROD10:%.*]] = fmul float [[IV10]], [[MV10]]
+; CHECK-NEXT: [[ACC10:%.*]] = fadd float [[ACC9]], [[PROD10]]
+; CHECK-NEXT: [[IP11:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 17
+; CHECK-NEXT: [[IV11:%.*]] = load float, ptr addrspace(1) [[IP11]], align 4
+; CHECK-NEXT: [[MP11:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 11
+; CHECK-NEXT: [[MV11:%.*]] = load float, ptr addrspace(4) [[MP11]], align 4
+; CHECK-NEXT: [[PROD11:%.*]] = fmul float [[IV11]], [[MV11]]
+; CHECK-NEXT: [[ACC11:%.*]] = fadd float [[ACC10]], [[PROD11]]
; CHECK-NEXT: [[IP12:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 18
-; CHECK-NEXT: [[IP14:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 20
-; CHECK-NEXT: [[IV14:%.*]] = load float, ptr addrspace(1) [[IP14]], align 4
+; CHECK-NEXT: [[IV12:%.*]] = load float, ptr addrspace(1) [[IP12]], align 4
+; CHECK-NEXT: [[MP12:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 12
+; CHECK-NEXT: [[MV12:%.*]] = load float, ptr addrspace(4) [[MP12]], align 4
+; CHECK-NEXT: [[PROD12:%.*]] = fmul float [[IV12]], [[MV12]]
+; CHECK-NEXT: [[ACC12:%.*]] = fadd float [[ACC11]], [[PROD12]]
+; CHECK-NEXT: [[IP13:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 19
+; CHECK-NEXT: [[MP13:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 13
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x float>, ptr addrspace(1) [[IP13]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = load <2 x float>, ptr addrspace(4) [[MP13]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = fmul <2 x float> [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x float> [[TMP3]], i64 0
+; CHECK-NEXT: [[ACC13:%.*]] = fadd float [[ACC12]], [[TMP4]]
+; CHECK-NEXT: [[TMP5:%.*]] = extractelement <2 x float> [[TMP3]], i64 1
+; CHECK-NEXT: [[ACC14:%.*]] = fadd float [[ACC13]], [[TMP5]]
; CHECK-NEXT: [[IP15:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 24
; CHECK-NEXT: [[IV15:%.*]] = load float, ptr addrspace(1) [[IP15]], align 4
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr addrspace(1) [[IP0]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = load <2 x float>, ptr addrspace(1) [[IP6]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = load <2 x float>, ptr addrspace(1) [[IP8]], align 4
-; CHECK-NEXT: [[TMP4:%.*]] = load <2 x float>, ptr addrspace(1) [[IP10]], align 4
-; CHECK-NEXT: [[TMP5:%.*]] = load <2 x float>, ptr addrspace(1) [[IP12]], align 4
-; CHECK-NEXT: [[TMP6:%.*]] = load <16 x float>, ptr addrspace(4) [[MP0]], align 4
-; CHECK-NEXT: [[TMP7:%.*]] = insertelement <16 x float> poison, float [[IV4]], i64 4
-; CHECK-NEXT: [[TMP8:%.*]] = insertelement <16 x float> [[TMP7]], float [[IV5]], i64 5
-; CHECK-NEXT: [[TMP9:%.*]] = insertelement <16 x float> [[TMP8]], float [[IV14]], i64 14
-; CHECK-NEXT: [[TMP10:%.*]] = insertelement <16 x float> [[TMP9]], float [[IV15]], i64 15
-; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <16 x float> [[TMP10]], <16 x float> [[TMP11]], <16 x i32> <i32 16, i32 17, i32 18, i32 19, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
-; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <2 x float> [[TMP2]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP14:%.*]] = shufflevector <16 x float> [[TMP12]], <16 x float> [[TMP13]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 16, i32 17, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
-; CHECK-NEXT: [[TMP15:%.*]] = shufflevector <2 x float> [[TMP3]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP16:%.*]] = shufflevector <16 x float> [[TMP14]], <16 x float> [[TMP15]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 16, i32 17, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
-; CHECK-NEXT: [[TMP17:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP18:%.*]] = shufflevector <16 x float> [[TMP16]], <16 x float> [[TMP17]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 16, i32 17, i32 12, i32 13, i32 14, i32 15>
-; CHECK-NEXT: [[TMP19:%.*]] = shufflevector <2 x float> [[TMP5]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP20:%.*]] = shufflevector <16 x float> [[TMP18]], <16 x float> [[TMP19]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 16, i32 17, i32 14, i32 15>
-; CHECK-NEXT: [[TMP21:%.*]] = fmul <16 x float> [[TMP20]], [[TMP6]]
-; CHECK-NEXT: [[TMP22:%.*]] = extractelement <16 x float> [[TMP21]], i64 0
-; CHECK-NEXT: [[TMP23:%.*]] = extractelement <16 x float> [[TMP21]], i64 1
-; CHECK-NEXT: [[ACC1:%.*]] = fadd float [[TMP22]], [[TMP23]]
-; CHECK-NEXT: [[TMP24:%.*]] = extractelement <16 x float> [[TMP21]], i64 2
-; CHECK-NEXT: [[ACC2:%.*]] = fadd float [[ACC1]], [[TMP24]]
-; CHECK-NEXT: [[TMP25:%.*]] = extractelement <16 x float> [[TMP21]], i64 3
-; CHECK-NEXT: [[ACC3:%.*]] = fadd float [[ACC2]], [[TMP25]]
-; CHECK-NEXT: [[TMP26:%.*]] = extractelement <16 x float> [[TMP21]], i64 4
-; CHECK-NEXT: [[ACC4:%.*]] = fadd float [[ACC3]], [[TMP26]]
-; CHECK-NEXT: [[TMP27:%.*]] = extractelement <16 x float> [[TMP21]], i64 5
-; CHECK-NEXT: [[ACC5:%.*]] = fadd float [[ACC4]], [[TMP27]]
-; CHECK-NEXT: [[TMP28:%.*]] = extractelement <16 x float> [[TMP21]], i64 6
-; CHECK-NEXT: [[ACC6:%.*]] = fadd float [[ACC5]], [[TMP28]]
-; CHECK-NEXT: [[TMP29:%.*]] = extractelement <16 x float> [[TMP21]], i64 7
-; CHECK-NEXT: [[ACC7:%.*]] = fadd float [[ACC6]], [[TMP29]]
-; CHECK-NEXT: [[TMP30:%.*]] = extractelement <16 x float> [[TMP21]], i64 8
-; CHECK-NEXT: [[ACC8:%.*]] = fadd float [[ACC7]], [[TMP30]]
-; CHECK-NEXT: [[TMP31:%.*]] = extractelement <16 x float> [[TMP21]], i64 9
-; CHECK-NEXT: [[ACC9:%.*]] = fadd float [[ACC8]], [[TMP31]]
-; CHECK-NEXT: [[TMP32:%.*]] = extractelement <16 x float> [[TMP21]], i64 10
-; CHECK-NEXT: [[ACC10:%.*]] = fadd float [[ACC9]], [[TMP32]]
-; CHECK-NEXT: [[TMP33:%.*]] = extractelement <16 x float> [[TMP21]], i64 11
-; CHECK-NEXT: [[ACC11:%.*]] = fadd float [[ACC10]], [[TMP33]]
-; CHECK-NEXT: [[TMP34:%.*]] = extractelement <16 x float> [[TMP21]], i64 12
-; CHECK-NEXT: [[ACC12:%.*]] = fadd float [[ACC11]], [[TMP34]]
-; CHECK-NEXT: [[TMP35:%.*]] = extractelement <16 x float> [[TMP21]], i64 13
-; CHECK-NEXT: [[ACC13:%.*]] = fadd float [[ACC12]], [[TMP35]]
-; CHECK-NEXT: [[TMP36:%.*]] = extractelement <16 x float> [[TMP21]], i64 14
-; CHECK-NEXT: [[ACC14:%.*]] = fadd float [[ACC13]], [[TMP36]]
-; CHECK-NEXT: [[TMP37:%.*]] = extractelement <16 x float> [[TMP21]], i64 15
+; CHECK-NEXT: [[MP15:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 15
+; CHECK-NEXT: [[MV15:%.*]] = load float, ptr addrspace(4) [[MP15]], align 4
+; CHECK-NEXT: [[TMP37:%.*]] = fmul float [[IV15]], [[MV15]]
; CHECK-NEXT: [[ACC15:%.*]] = fadd float [[ACC14]], [[TMP37]]
; CHECK-NEXT: [[IP16:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 25
; CHECK-NEXT: [[MP16:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 16
+; CHECK-NEXT: [[TMP6:%.*]] = load <2 x float>, ptr addrspace(1) [[IP16]], align 4
+; CHECK-NEXT: [[TMP7:%.*]] = load <2 x float>, ptr addrspace(4) [[MP16]], align 4
+; CHECK-NEXT: [[TMP8:%.*]] = fmul <2 x float> [[TMP6]], [[TMP7]]
+; CHECK-NEXT: [[TMP9:%.*]] = extractelement <2 x float> [[TMP8]], i64 0
+; CHECK-NEXT: [[ACC16:%.*]] = fadd float [[ACC15]], [[TMP9]]
+; CHECK-NEXT: [[TMP10:%.*]] = extractelement <2 x float> [[TMP8]], i64 1
+; CHECK-NEXT: [[ACC17:%.*]] = fadd float [[ACC16]], [[TMP10]]
+; CHECK-NEXT: [[IP18:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 27
+; CHECK-NEXT: [[MP18:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 18
+; CHECK-NEXT: [[TMP11:%.*]] = load <2 x float>, ptr addrspace(1) [[IP18]], align 4
+; CHECK-NEXT: [[TMP12:%.*]] = load <2 x float>, ptr addrspace(4) [[MP18]], align 4
+; CHECK-NEXT: [[TMP13:%.*]] = fmul <2 x float> [[TMP11]], [[TMP12]]
+; CHECK-NEXT: [[TMP14:%.*]] = extractelement <2 x float> [[TMP13]], i64 0
+; CHECK-NEXT: [[ACC18:%.*]] = fadd float [[ACC17]], [[TMP14]]
+; CHECK-NEXT: [[TMP15:%.*]] = extractelement <2 x float> [[TMP13]], i64 1
+; CHECK-NEXT: [[ACC19:%.*]] = fadd float [[ACC18]], [[TMP15]]
; CHECK-NEXT: [[IP20:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 32
-; CHECK-NEXT: [[TMP38:%.*]] = load <4 x float>, ptr addrspace(1) [[IP16]], align 4
-; CHECK-NEXT: [[TMP39:%.*]] = load <4 x float>, ptr addrspace(1) [[IP20]], align 4
-; CHECK-NEXT: [[TMP40:%.*]] = load <8 x float>, ptr addrspace(4) [[MP16]], align 4
-; CHECK-NEXT: [[TMP41:%.*]] = shufflevector <4 x float> [[TMP38]], <4 x float> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP42:%.*]] = shufflevector <4 x float> [[TMP39]], <4 x float> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP43:%.*]] = shufflevector <4 x float> [[TMP38]], <4 x float> [[TMP39]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; CHECK-NEXT: [[TMP44:%.*]] = fmul <8 x float> [[TMP43]], [[TMP40]]
-; CHECK-NEXT: [[TMP45:%.*]] = extractelement <8 x float> [[TMP44]], i64 0
-; CHECK-NEXT: [[ACC16:%.*]] = fadd float [[ACC15]], [[TMP45]]
-; CHECK-NEXT: [[TMP46:%.*]] = extractelement <8 x float> [[TMP44]], i64 1
-; CHECK-NEXT: [[ACC17:%.*]] = fadd float [[ACC16]], [[TMP46]]
-; CHECK-NEXT: [[TMP47:%.*]] = extractelement <8 x float> [[TMP44]], i64 2
-; CHECK-NEXT: [[ACC18:%.*]] = fadd float [[ACC17]], [[TMP47]]
-; CHECK-NEXT: [[TMP48:%.*]] = extractelement <8 x float> [[TMP44]], i64 3
-; CHECK-NEXT: [[ACC19:%.*]] = fadd float [[ACC18]], [[TMP48]]
-; CHECK-NEXT: [[TMP49:%.*]] = extractelement <8 x float> [[TMP44]], i64 4
-; CHECK-NEXT: [[ACC20:%.*]] = fadd float [[ACC19]], [[TMP49]]
-; CHECK-NEXT: [[TMP50:%.*]] = extractelement <8 x float> [[TMP44]], i64 5
-; CHECK-NEXT: [[ACC21:%.*]] = fadd float [[ACC20]], [[TMP50]]
-; CHECK-NEXT: [[TMP51:%.*]] = extractelement <8 x float> [[TMP44]], i64 6
-; CHECK-NEXT: [[ACC22:%.*]] = fadd float [[ACC21]], [[TMP51]]
-; CHECK-NEXT: [[TMP52:%.*]] = extractelement <8 x float> [[TMP44]], i64 7
-; CHECK-NEXT: [[ACC23:%.*]] = fadd float [[ACC22]], [[TMP52]]
-; CHECK-NEXT: [[IP24:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 36
-; CHECK-NEXT: [[IV24:%.*]] = load float, ptr addrspace(1) [[IP24]], align 4
-; CHECK-NEXT: [[MP24:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 24
+; CHECK-NEXT: [[IV24:%.*]] = load float, ptr addrspace(1) [[IP20]], align 4
+; CHECK-NEXT: [[MP24:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 20
; CHECK-NEXT: [[MV24:%.*]] = load float, ptr addrspace(4) [[MP24]], align 4
; CHECK-NEXT: [[PROD24:%.*]] = fmul float [[IV24]], [[MV24]]
-; CHECK-NEXT: [[ACC24:%.*]] = fadd float [[ACC23]], [[PROD24]]
+; CHECK-NEXT: [[ACC20:%.*]] = fadd float [[ACC19]], [[PROD24]]
+; CHECK-NEXT: [[IP21:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 33
+; CHECK-NEXT: [[MP21:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 21
+; CHECK-NEXT: [[TMP16:%.*]] = load <2 x float>, ptr addrspace(1) [[IP21]], align 4
+; CHECK-NEXT: [[TMP17:%.*]] = load <2 x float>, ptr addrspace(4) [[MP21]], align 4
+; CHECK-NEXT: [[TMP18:%.*]] = fmul <2 x float> [[TMP16]], [[TMP17]]
+; CHECK-NEXT: [[TMP19:%.*]] = extractelement <2 x float> [[TMP18]], i64 0
+; CHECK-NEXT: [[ACC21:%.*]] = fadd float [[ACC20]], [[TMP19]]
+; CHECK-NEXT: [[TMP20:%.*]] = extractelement <2 x float> [[TMP18]], i64 1
+; CHECK-NEXT: [[ACC22:%.*]] = fadd float [[ACC21]], [[TMP20]]
+; CHECK-NEXT: [[IP23:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 35
+; CHECK-NEXT: [[MP23:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 23
+; CHECK-NEXT: [[TMP21:%.*]] = load <2 x float>, ptr addrspace(1) [[IP23]], align 4
+; CHECK-NEXT: [[TMP22:%.*]] = load <2 x float>, ptr addrspace(4) [[MP23]], align 4
+; CHECK-NEXT: [[TMP23:%.*]] = fmul <2 x float> [[TMP21]], [[TMP22]]
+; CHECK-NEXT: [[TMP24:%.*]] = extractelement <2 x float> [[TMP23]], i64 0
+; CHECK-NEXT: [[ACC23:%.*]] = fadd float [[ACC22]], [[TMP24]]
+; CHECK-NEXT: [[TMP25:%.*]] = extractelement <2 x float> [[TMP23]], i64 1
+; CHECK-NEXT: [[ACC24:%.*]] = fadd float [[ACC23]], [[TMP25]]
; CHECK-NEXT: ret float [[ACC24]]
;
%ip0 = getelementptr inbounds float, ptr addrspace(1) %input, i64 0
>From a768a419d6ec0951d36a072a91e073b1755dd150 Mon Sep 17 00:00:00 2001
From: Akash Dutta <Akash.Dutta at amd.com>
Date: Thu, 20 Aug 2026 16:16:35 +0000
Subject: [PATCH 10/11] Cost only concrete packed-f32 InsertElements
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 37 +-
llvm/test/Analysis/CostModel/AMDGPU/cast.ll | 72 ++--
llvm/test/Analysis/CostModel/AMDGPU/fround.ll | 191 ++++++---
.../test/Analysis/CostModel/AMDGPU/maximum.ll | 93 ++---
llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll | 74 +---
.../test/Analysis/CostModel/AMDGPU/minimum.ll | 93 ++---
llvm/test/Analysis/CostModel/AMDGPU/minnum.ll | 74 +---
.../AMDGPU/packed-fp32-insertelement.ll | 140 +++++++
.../CostModel/AMDGPU/packed-fp32-shuffle.ll | 96 ++---
.../AMDGPU/ordered-reduction-fma-fusion.ll | 392 +++++++-----------
.../AMDGPU/combine-scalar-selects.ll | 43 +-
11 files changed, 614 insertions(+), 691 deletions(-)
create mode 100644 llvm/test/Analysis/CostModel/AMDGPU/packed-fp32-insertelement.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index edab6943370ca..27f9a52431688 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1046,17 +1046,20 @@ InstructionCost GCNTTIImpl::getVectorInstrCost(
VIC);
}
- // Gfx9 packed f32 pair formation for v_pk_*_f32: load-fed inserts are free;
- // compute-fed inserts cost scales with legalization (wide vectors split to
- // native <2 x f32>).
+ // Packed f32 pair formation for v_pk_*_f32. Only price a concrete insert
+ // (the inserted value Op1 is known): a load-fed lane is free, while a
+ // compute-fed lane costs the real packing work, scaled by legalization for
+ // wider vectors.
if (Opcode == Instruction::InsertElement && EltSize == 32 &&
- ST->hasPackedFP32Ops())
- if (auto *VecTy = dyn_cast<FixedVectorType>(ValTy))
+ ST->hasAnyPackedFP32Ops() && Op1) {
+ if (auto *VecTy = dyn_cast<FixedVectorType>(ValTy)) {
if (VecTy->getElementType()->isFloatTy()) {
- if (Op1 && isa<LoadInst>(Op1))
+ if (isa<LoadInst>(Op1))
return 0;
return getTypeLegalizationCost(ValTy).first;
}
+ }
+ }
// Extracts are just reads of a subregister, so are free. Inserts are
// considered free because we don't want to have any cost for scalarizing
@@ -1361,28 +1364,6 @@ InstructionCost GCNTTIImpl::getShuffleCost(TTI::ShuffleKind Kind,
Kind = improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTp);
unsigned ScalarSize = DL.getTypeSizeInBits(SrcTy->getElementType());
-
- // Legal <2 x f32> shuffles for v_pk_*_f32 sources. Identity and broadcasts
- // are free via op_sel on packed ops; high-to-low lane swap within an aligned
- // pair is lowered to v_pk_mov_b32.
- if (ST->hasPackedFP32Ops() && ScalarSize == 32) {
- auto *DstVecTy = dyn_cast<FixedVectorType>(DstTy);
- auto *SrcVecTy = dyn_cast<FixedVectorType>(SrcTy);
- if (DstVecTy && SrcVecTy && DstVecTy->getNumElements() == 2 &&
- SrcVecTy->getNumElements() == 2 &&
- DstVecTy->getElementType()->isFloatTy()) {
- switch (Kind) {
- case TTI::SK_Broadcast:
- case TTI::SK_PermuteSingleSrc:
- return 0;
- case TTI::SK_Reverse:
- return 1;
- default:
- break;
- }
- }
- }
-
if (ST->getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS &&
(ScalarSize == 16 || ScalarSize == 8)) {
// Larger vector widths may require additional instructions, but are
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/cast.ll b/llvm/test/Analysis/CostModel/AMDGPU/cast.ll
index 444e74577bc4b..65d1eacfbb611 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/cast.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/cast.ll
@@ -299,19 +299,19 @@ define void @sitofp4(<4 x i1> %a, <4 x i8> %b, <4 x i16> %c, <4 x i32> %d) {
}
define void @sitofp8(<8 x i1> %a, <8 x i8> %b, <8 x i16> %c, <8 x i32> %d) {
-; SLOW-LABEL: 'sitofp8'
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = sitofp <8 x i1> %a to <8 x float>
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = sitofp <8 x i8> %b to <8 x float>
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = sitofp <8 x i16> %c to <8 x float>
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = sitofp <8 x i32> %d to <8 x float>
-; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+; ALL-LABEL: 'sitofp8'
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = sitofp <8 x i1> %a to <8 x float>
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = sitofp <8 x i8> %b to <8 x float>
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = sitofp <8 x i16> %c to <8 x float>
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = sitofp <8 x i32> %d to <8 x float>
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; SLOW-SIZE-LABEL: 'sitofp8'
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = sitofp <8 x i1> %a to <8 x float>
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = sitofp <8 x i8> %b to <8 x float>
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = sitofp <8 x i16> %c to <8 x float>
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = sitofp <8 x i32> %d to <8 x float>
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; ALL-SIZE-LABEL: 'sitofp8'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = sitofp <8 x i1> %a to <8 x float>
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = sitofp <8 x i8> %b to <8 x float>
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = sitofp <8 x i16> %c to <8 x float>
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = sitofp <8 x i32> %d to <8 x float>
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%A1 = sitofp <8 x i1> %a to <8 x float>
%B1 = sitofp <8 x i8> %b to <8 x float>
@@ -377,19 +377,19 @@ define void @uitofp4(<4 x i1> %a, <4 x i8> %b, <4 x i16> %c, <4 x i32> %d) {
}
define void @uitofp8(<8 x i1> %a, <8 x i8> %b, <8 x i16> %c, <8 x i32> %d) {
-; SLOW-LABEL: 'uitofp8'
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = uitofp <8 x i1> %a to <8 x float>
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = uitofp <8 x i8> %b to <8 x float>
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = uitofp <8 x i16> %c to <8 x float>
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = uitofp <8 x i32> %d to <8 x float>
-; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+; ALL-LABEL: 'uitofp8'
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = uitofp <8 x i1> %a to <8 x float>
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = uitofp <8 x i8> %b to <8 x float>
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = uitofp <8 x i16> %c to <8 x float>
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = uitofp <8 x i32> %d to <8 x float>
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; SLOW-SIZE-LABEL: 'uitofp8'
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = uitofp <8 x i1> %a to <8 x float>
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = uitofp <8 x i8> %b to <8 x float>
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = uitofp <8 x i16> %c to <8 x float>
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = uitofp <8 x i32> %d to <8 x float>
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; ALL-SIZE-LABEL: 'uitofp8'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A1 = uitofp <8 x i1> %a to <8 x float>
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %B1 = uitofp <8 x i8> %b to <8 x float>
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %C1 = uitofp <8 x i16> %c to <8 x float>
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %D1 = uitofp <8 x i32> %d to <8 x float>
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%A1 = uitofp <8 x i1> %a to <8 x float>
%B1 = uitofp <8 x i8> %b to <8 x float>
@@ -399,19 +399,19 @@ define void @uitofp8(<8 x i1> %a, <8 x i8> %b, <8 x i16> %c, <8 x i32> %d) {
}
define void @fp_conv(<8 x float> %a, <16 x float>%b, <4 x float> %c) {
-; SLOW-LABEL: 'fp_conv'
-; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A1 = fpext <4 x float> %c to <4 x double>
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A2 = fpext <8 x float> %a to <8 x double>
-; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A3 = fptrunc <4 x double> undef to <4 x float>
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A4 = fptrunc <8 x double> undef to <8 x float>
-; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+; ALL-LABEL: 'fp_conv'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A1 = fpext <4 x float> %c to <4 x double>
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A2 = fpext <8 x float> %a to <8 x double>
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A3 = fptrunc <4 x double> undef to <4 x float>
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A4 = fptrunc <8 x double> undef to <8 x float>
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; SLOW-SIZE-LABEL: 'fp_conv'
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A1 = fpext <4 x float> %c to <4 x double>
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A2 = fpext <8 x float> %a to <8 x double>
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A3 = fptrunc <4 x double> undef to <4 x float>
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A4 = fptrunc <8 x double> undef to <8 x float>
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; ALL-SIZE-LABEL: 'fp_conv'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A1 = fpext <4 x float> %c to <4 x double>
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A2 = fpext <8 x float> %a to <8 x double>
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %A3 = fptrunc <4 x double> undef to <4 x float>
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %A4 = fptrunc <8 x double> undef to <8 x float>
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%A1 = fpext <4 x float> %c to <4 x double>
%A2 = fpext <8 x float> %a to <8 x double>
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/fround.ll b/llvm/test/Analysis/CostModel/AMDGPU/fround.ll
index d78c50d0221e4..90e3ff231b5fa 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/fround.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/fround.ll
@@ -11,6 +11,17 @@
; END.
define i32 @ceil(i32 %arg) {
+; FAST-LABEL: 'ceil'
+; FAST-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.ceil.f32(float undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.ceil.v4f32(<4 x float> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.ceil.v8f32(<8 x float> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.ceil.v16f32(<16 x float> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.ceil.f64(double undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.ceil.v2f64(<2 x double> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.ceil.v4f64(<4 x double> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.ceil.v8f64(<8 x double> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
+;
; SLOW-LABEL: 'ceil'
; SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.ceil.f32(float undef)
; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.ceil.v4f32(<4 x float> undef)
@@ -22,6 +33,17 @@ define i32 @ceil(i32 %arg) {
; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.ceil.v8f64(<8 x double> undef)
; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
+; FAST-SIZE-LABEL: 'ceil'
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.ceil.f32(float undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.ceil.v4f32(<4 x float> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.ceil.v8f32(<8 x float> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.ceil.v16f32(<16 x float> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.ceil.f64(double undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.ceil.v2f64(<2 x double> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.ceil.v4f64(<4 x double> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.ceil.v8f64(<8 x double> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
+;
; SLOW-SIZE-LABEL: 'ceil'
; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.ceil.f32(float undef)
; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.ceil.v4f32(<4 x float> undef)
@@ -47,27 +69,27 @@ define i32 @ceil(i32 %arg) {
}
define i32 @floor(i32 %arg) {
-; SLOW-LABEL: 'floor'
-; SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.floor.f32(float undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.floor.v4f32(<4 x float> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.floor.v8f32(<8 x float> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.floor.v16f32(<16 x float> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.floor.f64(double undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.floor.v2f64(<2 x double> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.floor.v4f64(<4 x double> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.floor.v8f64(<8 x double> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
+; ALL-LABEL: 'floor'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.floor.f32(float undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.floor.v4f32(<4 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.floor.v8f32(<8 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.floor.v16f32(<16 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.floor.f64(double undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.floor.v2f64(<2 x double> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.floor.v4f64(<4 x double> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.floor.v8f64(<8 x double> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
-; SLOW-SIZE-LABEL: 'floor'
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.floor.f32(float undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.floor.v4f32(<4 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.floor.v8f32(<8 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.floor.v16f32(<16 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.floor.f64(double undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.floor.v2f64(<2 x double> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.floor.v4f64(<4 x double> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.floor.v8f64(<8 x double> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
+; ALL-SIZE-LABEL: 'floor'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.floor.f32(float undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.floor.v4f32(<4 x float> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.floor.v8f32(<8 x float> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.floor.v16f32(<16 x float> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.floor.f64(double undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.floor.v2f64(<2 x double> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.floor.v4f64(<4 x double> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.floor.v8f64(<8 x double> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
;
%F32 = call float @llvm.floor.f32(float undef)
%V4F32 = call <4 x float> @llvm.floor.v4f32(<4 x float> undef)
@@ -83,27 +105,27 @@ define i32 @floor(i32 %arg) {
}
define i32 @nearbyint(i32 %arg) {
-; SLOW-LABEL: 'nearbyint'
-; SLOW-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.nearbyint.f32(float undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.nearbyint.v4f32(<4 x float> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.nearbyint.v8f32(<8 x float> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.nearbyint.v16f32(<16 x float> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.nearbyint.f64(double undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.nearbyint.v2f64(<2 x double> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.nearbyint.v4f64(<4 x double> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.nearbyint.v8f64(<8 x double> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
+; ALL-LABEL: 'nearbyint'
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.nearbyint.f32(float undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.nearbyint.v4f32(<4 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.nearbyint.v8f32(<8 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.nearbyint.v16f32(<16 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.nearbyint.f64(double undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.nearbyint.v2f64(<2 x double> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.nearbyint.v4f64(<4 x double> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.nearbyint.v8f64(<8 x double> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
-; SLOW-SIZE-LABEL: 'nearbyint'
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.nearbyint.f32(float undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.nearbyint.v4f32(<4 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.nearbyint.v8f32(<8 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.nearbyint.v16f32(<16 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.nearbyint.f64(double undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.nearbyint.v2f64(<2 x double> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.nearbyint.v4f64(<4 x double> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.nearbyint.v8f64(<8 x double> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
+; ALL-SIZE-LABEL: 'nearbyint'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.nearbyint.f32(float undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.nearbyint.v4f32(<4 x float> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.nearbyint.v8f32(<8 x float> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.nearbyint.v16f32(<16 x float> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.nearbyint.f64(double undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.nearbyint.v2f64(<2 x double> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.nearbyint.v4f64(<4 x double> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.nearbyint.v8f64(<8 x double> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
;
%F32 = call float @llvm.nearbyint.f32(float undef)
%V4F32 = call <4 x float> @llvm.nearbyint.v4f32(<4 x float> undef)
@@ -119,27 +141,27 @@ define i32 @nearbyint(i32 %arg) {
}
define i32 @rint(i32 %arg) {
-; SLOW-LABEL: 'rint'
-; SLOW-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.rint.f32(float undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.rint.v4f32(<4 x float> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.rint.v8f32(<8 x float> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.rint.v16f32(<16 x float> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.rint.f64(double undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.rint.v2f64(<2 x double> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.rint.v4f64(<4 x double> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.rint.v8f64(<8 x double> undef)
-; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
+; ALL-LABEL: 'rint'
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.rint.f32(float undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.rint.v4f32(<4 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.rint.v8f32(<8 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.rint.v16f32(<16 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.rint.f64(double undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.rint.v2f64(<2 x double> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.rint.v4f64(<4 x double> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.rint.v8f64(<8 x double> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
-; SLOW-SIZE-LABEL: 'rint'
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.rint.f32(float undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.rint.v4f32(<4 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.rint.v8f32(<8 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.rint.v16f32(<16 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.rint.f64(double undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.rint.v2f64(<2 x double> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.rint.v4f64(<4 x double> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.rint.v8f64(<8 x double> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
+; ALL-SIZE-LABEL: 'rint'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.rint.f32(float undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F32 = call <4 x float> @llvm.rint.v4f32(<4 x float> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F32 = call <8 x float> @llvm.rint.v8f32(<8 x float> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %V16F32 = call <16 x float> @llvm.rint.v16f32(<16 x float> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.rint.f64(double undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V2F64 = call <2 x double> @llvm.rint.v2f64(<2 x double> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V4F64 = call <4 x double> @llvm.rint.v4f64(<4 x double> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.rint.v8f64(<8 x double> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
;
%F32 = call float @llvm.rint.f32(float undef)
%V4F32 = call <4 x float> @llvm.rint.v4f32(<4 x float> undef)
@@ -155,6 +177,17 @@ define i32 @rint(i32 %arg) {
}
define i32 @roundeven(i32 %arg) {
+; FAST-LABEL: 'roundeven'
+; FAST-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.roundeven.f32(float undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.roundeven.v4f32(<4 x float> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.roundeven.v8f32(<8 x float> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.roundeven.v16f32(<16 x float> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.roundeven.f64(double undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.roundeven.v2f64(<2 x double> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.roundeven.v4f64(<4 x double> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.roundeven.v8f64(<8 x double> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
+;
; SLOW-LABEL: 'roundeven'
; SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.roundeven.f32(float undef)
; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.roundeven.v4f32(<4 x float> undef)
@@ -166,6 +199,17 @@ define i32 @roundeven(i32 %arg) {
; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.roundeven.v8f64(<8 x double> undef)
; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
+; FAST-SIZE-LABEL: 'roundeven'
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.roundeven.f32(float undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.roundeven.v4f32(<4 x float> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.roundeven.v8f32(<8 x float> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.roundeven.v16f32(<16 x float> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.roundeven.f64(double undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.roundeven.v2f64(<2 x double> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.roundeven.v4f64(<4 x double> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.roundeven.v8f64(<8 x double> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
+;
; SLOW-SIZE-LABEL: 'roundeven'
; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.roundeven.f32(float undef)
; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.roundeven.v4f32(<4 x float> undef)
@@ -191,6 +235,17 @@ define i32 @roundeven(i32 %arg) {
}
define i32 @trunc(i32 %arg) {
+; FAST-LABEL: 'trunc'
+; FAST-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.trunc.f32(float undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.trunc.v4f32(<4 x float> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.trunc.v8f32(<8 x float> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.trunc.v16f32(<16 x float> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.trunc.f64(double undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.trunc.v2f64(<2 x double> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.trunc.v4f64(<4 x double> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.trunc.v8f64(<8 x double> undef)
+; FAST-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
+;
; SLOW-LABEL: 'trunc'
; SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.trunc.f32(float undef)
; SLOW-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.trunc.v4f32(<4 x float> undef)
@@ -202,6 +257,17 @@ define i32 @trunc(i32 %arg) {
; SLOW-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V8F64 = call <8 x double> @llvm.trunc.v8f64(<8 x double> undef)
; SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
+; FAST-SIZE-LABEL: 'trunc'
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.trunc.f32(float undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.trunc.v4f32(<4 x float> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F32 = call <8 x float> @llvm.trunc.v8f32(<8 x float> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %V16F32 = call <16 x float> @llvm.trunc.v16f32(<16 x float> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F64 = call double @llvm.trunc.f64(double undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.trunc.v2f64(<2 x double> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.trunc.v4f64(<4 x double> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.trunc.v8f64(<8 x double> undef)
+; FAST-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
+;
; SLOW-SIZE-LABEL: 'trunc'
; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %F32 = call float @llvm.trunc.f32(float undef)
; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F32 = call <4 x float> @llvm.trunc.v4f32(<4 x float> undef)
@@ -285,8 +351,3 @@ declare double @llvm.trunc.f64(double)
declare <2 x double> @llvm.trunc.v2f64(<2 x double>)
declare <4 x double> @llvm.trunc.v4f64(<4 x double>)
declare <8 x double> @llvm.trunc.v8f64(<8 x double>)
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; ALL: {{.*}}
-; ALL-SIZE: {{.*}}
-; FAST: {{.*}}
-; FAST-SIZE: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/maximum.ll b/llvm/test/Analysis/CostModel/AMDGPU/maximum.ll
index 303b9a3ed65a2..935923b7f8213 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/maximum.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/maximum.ll
@@ -5,7 +5,7 @@
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu6.01-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=ALL,SLOWF64 %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.50-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX950-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.0a-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX90A-SIZE %s
-; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX900-SIZE %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu6.01-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,SLOW-SIZE %s
define void @maximum_f16() {
@@ -155,75 +155,30 @@ define void @maximum_bf16() {
define void @maximum_f32() {
; GFX950-FASTF64-LABEL: 'maximum_f32'
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 64 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; GFX90A-FASTF64-LABEL: 'maximum_f32'
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 22 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 44 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 88 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 208 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; FASTF64-LABEL: 'maximum_f32'
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; SLOWF64-LABEL: 'maximum_f32'
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; GFX950-SIZE-LABEL: 'maximum_f32'
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 64 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; GFX90A-SIZE-LABEL: 'maximum_f32'
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 64 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; GFX900-SIZE-LABEL: 'maximum_f32'
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; ALL-LABEL: 'maximum_f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; SLOW-SIZE-LABEL: 'maximum_f32'
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; SIZE-LABEL: 'maximum_f32'
+; SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.maximum.f32(float undef, float undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.maximum.v3f32(<3 x float> undef, <3 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.maximum.v4f32(<4 x float> undef, <4 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.maximum.v8f32(<8 x float> undef, <8 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.maximum.v16f32(<16 x float> undef, <16 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%f32 = call float @llvm.maximum.f32(float undef, float undef)
%v2f32 = call <2 x float> @llvm.maximum.v2f32(<2 x float> undef, <2 x float> undef)
@@ -270,3 +225,7 @@ define void @maximum_f64() {
%v16f64 = call <16 x double> @llvm.maximum.v16f64(<16 x double> undef, <16 x double> undef)
ret void
}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; FASTF64: {{.*}}
+; GFX90A-FASTF64: {{.*}}
+; GFX90A-SIZE: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll b/llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll
index 58962c0ed8cda..1acdfe972a4a2 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/maxnum.ll
@@ -3,7 +3,7 @@
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=ALL,GFX9,FASTF64 %s
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu6.01-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=ALL,SLOWF64 %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.0a-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX90A-SIZE %s
-; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX900-SIZE %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu6.01-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,SLOW-SIZE %s
define void @maxnum_f16() {
@@ -115,59 +115,23 @@ define void @maxnum_bf16() {
}
define void @maxnum_f32() {
-; GFX90A-FASTF64-LABEL: 'maxnum_f32'
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 12 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 24 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; FASTF64-LABEL: 'maxnum_f32'
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; SLOWF64-LABEL: 'maxnum_f32'
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; GFX90A-SIZE-LABEL: 'maxnum_f32'
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 12 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 24 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; GFX900-SIZE-LABEL: 'maxnum_f32'
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; ALL-LABEL: 'maxnum_f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; SLOW-SIZE-LABEL: 'maxnum_f32'
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; SIZE-LABEL: 'maxnum_f32'
+; SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.maxnum.f32(float undef, float undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.maxnum.v3f32(<3 x float> undef, <3 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.maxnum.v4f32(<4 x float> undef, <4 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.maxnum.v8f32(<8 x float> undef, <8 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.maxnum.v16f32(<16 x float> undef, <16 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%f32 = call float @llvm.maxnum.f32(float undef, float undef)
%v2f32 = call <2 x float> @llvm.maxnum.v2f32(<2 x float> undef, <2 x float> undef)
@@ -205,3 +169,7 @@ define void @maxnum_f64() {
%v16f64 = call <16 x double> @llvm.maxnum.v16f64(<16 x double> undef, <16 x double> undef)
ret void
}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; FASTF64: {{.*}}
+; GFX90A-FASTF64: {{.*}}
+; GFX90A-SIZE: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/minimum.ll b/llvm/test/Analysis/CostModel/AMDGPU/minimum.ll
index 4995487f5e225..116c28933fffa 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/minimum.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/minimum.ll
@@ -5,7 +5,7 @@
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu6.01-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=ALL,SLOWF64 %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.50-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX950-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.0a-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX90A-SIZE %s
-; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX900-SIZE %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu6.01-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,SLOW-SIZE %s
define void @minimum_f16() {
@@ -155,75 +155,30 @@ define void @minimum_bf16() {
define void @minimum_f32() {
; GFX950-FASTF64-LABEL: 'minimum_f32'
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 64 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
; GFX950-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; GFX90A-FASTF64-LABEL: 'minimum_f32'
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 22 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 44 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 88 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 208 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; FASTF64-LABEL: 'minimum_f32'
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; SLOWF64-LABEL: 'minimum_f32'
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; GFX950-SIZE-LABEL: 'minimum_f32'
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 64 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX950-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; GFX90A-SIZE-LABEL: 'minimum_f32'
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 64 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; GFX900-SIZE-LABEL: 'minimum_f32'
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; ALL-LABEL: 'minimum_f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 30 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 160 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; SLOW-SIZE-LABEL: 'minimum_f32'
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; SIZE-LABEL: 'minimum_f32'
+; SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f32 = call float @llvm.minimum.f32(float undef, float undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v3f32 = call <3 x float> @llvm.minimum.v3f32(<3 x float> undef, <3 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v4f32 = call <4 x float> @llvm.minimum.v4f32(<4 x float> undef, <4 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v8f32 = call <8 x float> @llvm.minimum.v8f32(<8 x float> undef, <8 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v16f32 = call <16 x float> @llvm.minimum.v16f32(<16 x float> undef, <16 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%f32 = call float @llvm.minimum.f32(float undef, float undef)
%v2f32 = call <2 x float> @llvm.minimum.v2f32(<2 x float> undef, <2 x float> undef)
@@ -270,3 +225,7 @@ define void @minimum_f64() {
%v16f64 = call <16 x double> @llvm.minimum.v16f64(<16 x double> undef, <16 x double> undef)
ret void
}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; FASTF64: {{.*}}
+; GFX90A-FASTF64: {{.*}}
+; GFX90A-SIZE: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/minnum.ll b/llvm/test/Analysis/CostModel/AMDGPU/minnum.ll
index baabdfd847fbf..121db02159228 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/minnum.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/minnum.ll
@@ -3,7 +3,7 @@
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=ALL,GFX9,FASTF64 %s
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu6.01-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=ALL,SLOWF64 %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.0a-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX90A-SIZE %s
-; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE,GFX900-SIZE %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -mattr=+half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,GFX9-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu6.01-unknown-amdhsa -mattr=-half-rate-64-ops < %s | FileCheck -check-prefixes=SIZE,SLOW-SIZE %s
define void @minnum_f16() {
@@ -115,59 +115,23 @@ define void @minnum_bf16() {
}
define void @minnum_f32() {
-; GFX90A-FASTF64-LABEL: 'minnum_f32'
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 12 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 24 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX90A-FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; FASTF64-LABEL: 'minnum_f32'
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
-; FASTF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; SLOWF64-LABEL: 'minnum_f32'
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
-; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; GFX90A-SIZE-LABEL: 'minnum_f32'
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 12 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 24 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; GFX900-SIZE-LABEL: 'minnum_f32'
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; ALL-LABEL: 'minnum_f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; SLOW-SIZE-LABEL: 'minnum_f32'
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
-; SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; SIZE-LABEL: 'minnum_f32'
+; SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %f32 = call float @llvm.minnum.f32(float undef, float undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %v3f32 = call <3 x float> @llvm.minnum.v3f32(<3 x float> undef, <3 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v4f32 = call <4 x float> @llvm.minnum.v4f32(<4 x float> undef, <4 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %v8f32 = call <8 x float> @llvm.minnum.v8f32(<8 x float> undef, <8 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 32 for instruction: %v16f32 = call <16 x float> @llvm.minnum.v16f32(<16 x float> undef, <16 x float> undef)
+; SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%f32 = call float @llvm.minnum.f32(float undef, float undef)
%v2f32 = call <2 x float> @llvm.minnum.v2f32(<2 x float> undef, <2 x float> undef)
@@ -205,3 +169,7 @@ define void @minnum_f64() {
%v16f64 = call <16 x double> @llvm.minnum.v16f64(<16 x double> undef, <16 x double> undef)
ret void
}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; FASTF64: {{.*}}
+; GFX90A-FASTF64: {{.*}}
+; GFX90A-SIZE: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/packed-fp32-insertelement.ll b/llvm/test/Analysis/CostModel/AMDGPU/packed-fp32-insertelement.ll
new file mode 100644
index 0000000000000..d8afe9bb80c62
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/AMDGPU/packed-fp32-insertelement.ll
@@ -0,0 +1,140 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.0a-unknown-amdhsa -S | FileCheck -check-prefixes=GFX90A %s
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -S | FileCheck -check-prefixes=GFX900 %s
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.0a-unknown-amdhsa -cost-kind=code-size -S | FileCheck -check-prefixes=GFX90A-SIZE %s
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.00-unknown-amdhsa -cost-kind=code-size -S | FileCheck -check-prefixes=GFX900-SIZE %s
+; END.
+
+; Packed-f32 pair-formation cost for v_pk_*_f32. On targets with packed FP32 ops
+; a compute-fed <N x float> insert is priced at the type-legalization cost (the
+; real packing work), while a load-fed lane stays free. On targets without
+; packed FP32 (GFX900) all inserts remain free, and i32 inserts are unaffected
+; on every target.
+
+define void @insert_f32_load_fed(ptr %p, ptr %q) {
+; GFX90A-LABEL: 'insert_f32_load_fed'
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %l0 = load float, ptr %p, align 4
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %l1 = load float, ptr %q, align 4
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2 = insertelement <2 x float> poison, float %l0, i32 0
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v4 = insertelement <4 x float> poison, float %l0, i32 1
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v8 = insertelement <8 x float> poison, float %l1, i32 3
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v16 = insertelement <16 x float> poison, float %l1, i32 7
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; GFX900-LABEL: 'insert_f32_load_fed'
+; GFX900-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %l0 = load float, ptr %p, align 4
+; GFX900-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %l1 = load float, ptr %q, align 4
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2 = insertelement <2 x float> poison, float %l0, i32 0
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v4 = insertelement <4 x float> poison, float %l0, i32 1
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v8 = insertelement <8 x float> poison, float %l1, i32 3
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v16 = insertelement <16 x float> poison, float %l1, i32 7
+; GFX900-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; GFX90A-SIZE-LABEL: 'insert_f32_load_fed'
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %l0 = load float, ptr %p, align 4
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %l1 = load float, ptr %q, align 4
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2 = insertelement <2 x float> poison, float %l0, i32 0
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v4 = insertelement <4 x float> poison, float %l0, i32 1
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v8 = insertelement <8 x float> poison, float %l1, i32 3
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v16 = insertelement <16 x float> poison, float %l1, i32 7
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; GFX900-SIZE-LABEL: 'insert_f32_load_fed'
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %l0 = load float, ptr %p, align 4
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %l1 = load float, ptr %q, align 4
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2 = insertelement <2 x float> poison, float %l0, i32 0
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v4 = insertelement <4 x float> poison, float %l0, i32 1
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v8 = insertelement <8 x float> poison, float %l1, i32 3
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v16 = insertelement <16 x float> poison, float %l1, i32 7
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %l0 = load float, ptr %p
+ %l1 = load float, ptr %q
+ %v2 = insertelement <2 x float> poison, float %l0, i32 0
+ %v4 = insertelement <4 x float> poison, float %l0, i32 1
+ %v8 = insertelement <8 x float> poison, float %l1, i32 3
+ %v16 = insertelement <16 x float> poison, float %l1, i32 7
+ ret void
+}
+
+define void @insert_f32_compute_fed(float %a, float %b) {
+; GFX90A-LABEL: 'insert_f32_compute_fed'
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %c = fadd float %a, %b
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2 = insertelement <2 x float> poison, float %c, i32 0
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v4 = insertelement <4 x float> poison, float %c, i32 1
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v8 = insertelement <8 x float> poison, float %c, i32 3
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v16 = insertelement <16 x float> poison, float %c, i32 7
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; GFX900-LABEL: 'insert_f32_compute_fed'
+; GFX900-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %c = fadd float %a, %b
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2 = insertelement <2 x float> poison, float %c, i32 0
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v4 = insertelement <4 x float> poison, float %c, i32 1
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v8 = insertelement <8 x float> poison, float %c, i32 3
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v16 = insertelement <16 x float> poison, float %c, i32 7
+; GFX900-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; GFX90A-SIZE-LABEL: 'insert_f32_compute_fed'
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %c = fadd float %a, %b
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2 = insertelement <2 x float> poison, float %c, i32 0
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v4 = insertelement <4 x float> poison, float %c, i32 1
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v8 = insertelement <8 x float> poison, float %c, i32 3
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %v16 = insertelement <16 x float> poison, float %c, i32 7
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; GFX900-SIZE-LABEL: 'insert_f32_compute_fed'
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %c = fadd float %a, %b
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2 = insertelement <2 x float> poison, float %c, i32 0
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v4 = insertelement <4 x float> poison, float %c, i32 1
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v8 = insertelement <8 x float> poison, float %c, i32 3
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v16 = insertelement <16 x float> poison, float %c, i32 7
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %c = fadd float %a, %b
+ %v2 = insertelement <2 x float> poison, float %c, i32 0
+ %v4 = insertelement <4 x float> poison, float %c, i32 1
+ %v8 = insertelement <8 x float> poison, float %c, i32 3
+ %v16 = insertelement <16 x float> poison, float %c, i32 7
+ ret void
+}
+
+define void @insert_i32_compute_fed(i32 %a, i32 %b) {
+; GFX90A-LABEL: 'insert_i32_compute_fed'
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %c = add i32 %a, %b
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2 = insertelement <2 x i32> poison, i32 %c, i32 0
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v4 = insertelement <4 x i32> poison, i32 %c, i32 1
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v8 = insertelement <8 x i32> poison, i32 %c, i32 3
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v16 = insertelement <16 x i32> poison, i32 %c, i32 7
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; GFX900-LABEL: 'insert_i32_compute_fed'
+; GFX900-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %c = add i32 %a, %b
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2 = insertelement <2 x i32> poison, i32 %c, i32 0
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v4 = insertelement <4 x i32> poison, i32 %c, i32 1
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v8 = insertelement <8 x i32> poison, i32 %c, i32 3
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v16 = insertelement <16 x i32> poison, i32 %c, i32 7
+; GFX900-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; GFX90A-SIZE-LABEL: 'insert_i32_compute_fed'
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %c = add i32 %a, %b
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2 = insertelement <2 x i32> poison, i32 %c, i32 0
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v4 = insertelement <4 x i32> poison, i32 %c, i32 1
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v8 = insertelement <8 x i32> poison, i32 %c, i32 3
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v16 = insertelement <16 x i32> poison, i32 %c, i32 7
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; GFX900-SIZE-LABEL: 'insert_i32_compute_fed'
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %c = add i32 %a, %b
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2 = insertelement <2 x i32> poison, i32 %c, i32 0
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v4 = insertelement <4 x i32> poison, i32 %c, i32 1
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v8 = insertelement <8 x i32> poison, i32 %c, i32 3
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v16 = insertelement <16 x i32> poison, i32 %c, i32 7
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %c = add i32 %a, %b
+ %v2 = insertelement <2 x i32> poison, i32 %c, i32 0
+ %v4 = insertelement <4 x i32> poison, i32 %c, i32 1
+ %v8 = insertelement <8 x i32> poison, i32 %c, i32 3
+ %v16 = insertelement <16 x i32> poison, i32 %c, i32 7
+ ret void
+}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/packed-fp32-shuffle.ll b/llvm/test/Analysis/CostModel/AMDGPU/packed-fp32-shuffle.ll
index 949275aff9798..376b60f2eb0ee 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/packed-fp32-shuffle.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/packed-fp32-shuffle.ll
@@ -7,83 +7,49 @@
; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu8.03-unknown-amdhsa -cost-kind=code-size -S | FileCheck -check-prefixes=VI-SIZE %s
; END.
-; Costs for legal <2 x f32> shuffles used to form v_pk_*_f32 sources. Identity and
-; broadcasts are free via op_sel; high-to-low lane swap needs v_pk_mov_b32.
+; Wider f32 shuffles must not inherit InsertElement legalization costs from the
+; packed-f32 pair-formation model: a splat of an f32 vector is not a scalarized
+; buildvector, so it stays free rather than being priced as N inserts.
-define amdgpu_kernel void @packed_fp32_shufflevector(<2 x float> %vec1, <2 x float> %vec2) {
-; GFX90A-LABEL: 'packed_fp32_shufflevector'
-; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
-; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
-; GFX90A-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
-; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
-; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
-; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
-; GFX90A-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
-; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+define amdgpu_kernel void @packed_fp32_wide_shufflevector(<4 x float> %vec4, <8 x float> %vec8, <16 x float> %vec16) {
+; GFX90A-LABEL: 'packed_fp32_wide_shufflevector'
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf4 = shufflevector <4 x float> %vec4, <4 x float> %vec4, <4 x i32> zeroinitializer
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf8 = shufflevector <8 x float> %vec8, <8 x float> %vec8, <8 x i32> zeroinitializer
+; GFX90A-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf16 = shufflevector <16 x float> %vec16, <16 x float> %vec16, <16 x i32> zeroinitializer
; GFX90A-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; GFX900-LABEL: 'packed_fp32_shufflevector'
-; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
-; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
-; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
-; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
-; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
-; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
-; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
-; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+; GFX900-LABEL: 'packed_fp32_wide_shufflevector'
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf4 = shufflevector <4 x float> %vec4, <4 x float> %vec4, <4 x i32> zeroinitializer
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf8 = shufflevector <8 x float> %vec8, <8 x float> %vec8, <8 x i32> zeroinitializer
+; GFX900-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf16 = shufflevector <16 x float> %vec16, <16 x float> %vec16, <16 x i32> zeroinitializer
; GFX900-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; VI-LABEL: 'packed_fp32_shufflevector'
-; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
-; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
-; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
-; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
-; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
-; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
-; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
-; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+; VI-LABEL: 'packed_fp32_wide_shufflevector'
+; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf4 = shufflevector <4 x float> %vec4, <4 x float> %vec4, <4 x i32> zeroinitializer
+; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf8 = shufflevector <8 x float> %vec8, <8 x float> %vec8, <8 x i32> zeroinitializer
+; VI-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf16 = shufflevector <16 x float> %vec16, <16 x float> %vec16, <16 x i32> zeroinitializer
; VI-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; GFX90A-SIZE-LABEL: 'packed_fp32_shufflevector'
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
-; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+; GFX90A-SIZE-LABEL: 'packed_fp32_wide_shufflevector'
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf4 = shufflevector <4 x float> %vec4, <4 x float> %vec4, <4 x i32> zeroinitializer
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf8 = shufflevector <8 x float> %vec8, <8 x float> %vec8, <8 x i32> zeroinitializer
+; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf16 = shufflevector <16 x float> %vec16, <16 x float> %vec16, <16 x i32> zeroinitializer
; GFX90A-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; GFX900-SIZE-LABEL: 'packed_fp32_shufflevector'
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
-; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+; GFX900-SIZE-LABEL: 'packed_fp32_wide_shufflevector'
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf4 = shufflevector <4 x float> %vec4, <4 x float> %vec4, <4 x i32> zeroinitializer
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf8 = shufflevector <8 x float> %vec8, <8 x float> %vec8, <8 x i32> zeroinitializer
+; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf16 = shufflevector <16 x float> %vec16, <16 x float> %vec16, <16 x i32> zeroinitializer
; GFX900-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; VI-SIZE-LABEL: 'packed_fp32_shufflevector'
-; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
-; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
-; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
-; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
-; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
-; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
-; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
-; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+; VI-SIZE-LABEL: 'packed_fp32_wide_shufflevector'
+; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf4 = shufflevector <4 x float> %vec4, <4 x float> %vec4, <4 x i32> zeroinitializer
+; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf8 = shufflevector <8 x float> %vec8, <8 x float> %vec8, <8 x i32> zeroinitializer
+; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %shuf16 = shufflevector <16 x float> %vec16, <16 x float> %vec16, <16 x i32> zeroinitializer
; VI-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- %shuf00 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> zeroinitializer
- %shuf01 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 0, i32 1>
- %shuf10 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 0>
- %shuf11 = shufflevector <2 x float> %vec1, <2 x float> %vec1, <2 x i32> <i32 1, i32 1>
- %shuf00_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> zeroinitializer
- %shuf01_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 0, i32 1>
- %shuf10_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 0>
- %shuf11_2 = shufflevector <2 x float> %vec1, <2 x float> %vec2, <2 x i32> <i32 1, i32 1>
+ %shuf4 = shufflevector <4 x float> %vec4, <4 x float> %vec4, <4 x i32> zeroinitializer
+ %shuf8 = shufflevector <8 x float> %vec8, <8 x float> %vec8, <8 x i32> zeroinitializer
+ %shuf16 = shufflevector <16 x float> %vec16, <16 x float> %vec16, <16 x i32> zeroinitializer
ret void
}
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/ordered-reduction-fma-fusion.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/ordered-reduction-fma-fusion.ll
index 957c7a532d590..6a7a93a44f8c9 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/ordered-reduction-fma-fusion.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/ordered-reduction-fma-fusion.ll
@@ -9,139 +9,103 @@
define float @conv_contract(ptr addrspace(1) %input, ptr addrspace(4) %mask) {
; CHECK-LABEL: @conv_contract(
; CHECK-NEXT: [[IP0:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT:%.*]], i64 0
-; CHECK-NEXT: [[IV0:%.*]] = load float, ptr addrspace(1) [[IP0]], align 4
; CHECK-NEXT: [[MP0:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK:%.*]], i64 0
-; CHECK-NEXT: [[MV0:%.*]] = load float, ptr addrspace(4) [[MP0]], align 4
-; CHECK-NEXT: [[PROD0:%.*]] = fmul contract float [[IV0]], [[MV0]]
-; CHECK-NEXT: [[IP1:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 1
-; CHECK-NEXT: [[IV1:%.*]] = load float, ptr addrspace(1) [[IP1]], align 4
-; CHECK-NEXT: [[MP1:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 1
-; CHECK-NEXT: [[MV1:%.*]] = load float, ptr addrspace(4) [[MP1]], align 4
-; CHECK-NEXT: [[PROD1:%.*]] = fmul contract float [[IV1]], [[MV1]]
-; CHECK-NEXT: [[ACC1:%.*]] = fadd contract float [[PROD0]], [[PROD1]]
-; CHECK-NEXT: [[IP2:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 2
-; CHECK-NEXT: [[IV2:%.*]] = load float, ptr addrspace(1) [[IP2]], align 4
-; CHECK-NEXT: [[MP2:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 2
-; CHECK-NEXT: [[MV2:%.*]] = load float, ptr addrspace(4) [[MP2]], align 4
-; CHECK-NEXT: [[PROD2:%.*]] = fmul contract float [[IV2]], [[MV2]]
-; CHECK-NEXT: [[ACC2:%.*]] = fadd contract float [[ACC1]], [[PROD2]]
-; CHECK-NEXT: [[IP3:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 3
-; CHECK-NEXT: [[IV3:%.*]] = load float, ptr addrspace(1) [[IP3]], align 4
-; CHECK-NEXT: [[MP3:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 3
-; CHECK-NEXT: [[MV3:%.*]] = load float, ptr addrspace(4) [[MP3]], align 4
-; CHECK-NEXT: [[PROD3:%.*]] = fmul contract float [[IV3]], [[MV3]]
-; CHECK-NEXT: [[ACC3:%.*]] = fadd contract float [[ACC2]], [[PROD3]]
; CHECK-NEXT: [[IP4:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 4
; CHECK-NEXT: [[IV4:%.*]] = load float, ptr addrspace(1) [[IP4]], align 4
-; CHECK-NEXT: [[MP4:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 4
-; CHECK-NEXT: [[MV4:%.*]] = load float, ptr addrspace(4) [[MP4]], align 4
-; CHECK-NEXT: [[PROD4:%.*]] = fmul contract float [[IV4]], [[MV4]]
-; CHECK-NEXT: [[ACC4:%.*]] = fadd contract float [[ACC3]], [[PROD4]]
; CHECK-NEXT: [[IP5:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 8
; CHECK-NEXT: [[IV5:%.*]] = load float, ptr addrspace(1) [[IP5]], align 4
-; CHECK-NEXT: [[MP5:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 5
-; CHECK-NEXT: [[MV5:%.*]] = load float, ptr addrspace(4) [[MP5]], align 4
-; CHECK-NEXT: [[PROD5:%.*]] = fmul contract float [[IV5]], [[MV5]]
-; CHECK-NEXT: [[ACC5:%.*]] = fadd contract float [[ACC4]], [[PROD5]]
; CHECK-NEXT: [[IP6:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 9
-; CHECK-NEXT: [[IV6:%.*]] = load float, ptr addrspace(1) [[IP6]], align 4
-; CHECK-NEXT: [[MP6:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 6
-; CHECK-NEXT: [[MV6:%.*]] = load float, ptr addrspace(4) [[MP6]], align 4
-; CHECK-NEXT: [[PROD6:%.*]] = fmul contract float [[IV6]], [[MV6]]
-; CHECK-NEXT: [[ACC6:%.*]] = fadd contract float [[ACC5]], [[PROD6]]
-; CHECK-NEXT: [[IP7:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 10
-; CHECK-NEXT: [[IV7:%.*]] = load float, ptr addrspace(1) [[IP7]], align 4
-; CHECK-NEXT: [[MP7:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 7
-; CHECK-NEXT: [[MV7:%.*]] = load float, ptr addrspace(4) [[MP7]], align 4
-; CHECK-NEXT: [[PROD7:%.*]] = fmul contract float [[IV7]], [[MV7]]
-; CHECK-NEXT: [[ACC7:%.*]] = fadd contract float [[ACC6]], [[PROD7]]
; CHECK-NEXT: [[IP8:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 11
-; CHECK-NEXT: [[IV8:%.*]] = load float, ptr addrspace(1) [[IP8]], align 4
-; CHECK-NEXT: [[MP8:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 8
-; CHECK-NEXT: [[MV8:%.*]] = load float, ptr addrspace(4) [[MP8]], align 4
-; CHECK-NEXT: [[PROD8:%.*]] = fmul contract float [[IV8]], [[MV8]]
-; CHECK-NEXT: [[ACC8:%.*]] = fadd contract float [[ACC7]], [[PROD8]]
-; CHECK-NEXT: [[IP9:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 12
-; CHECK-NEXT: [[IV9:%.*]] = load float, ptr addrspace(1) [[IP9]], align 4
-; CHECK-NEXT: [[MP9:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 9
-; CHECK-NEXT: [[MV9:%.*]] = load float, ptr addrspace(4) [[MP9]], align 4
-; CHECK-NEXT: [[PROD9:%.*]] = fmul contract float [[IV9]], [[MV9]]
-; CHECK-NEXT: [[ACC9:%.*]] = fadd contract float [[ACC8]], [[PROD9]]
; CHECK-NEXT: [[IP10:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 16
-; CHECK-NEXT: [[IV10:%.*]] = load float, ptr addrspace(1) [[IP10]], align 4
-; CHECK-NEXT: [[MP10:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 10
-; CHECK-NEXT: [[MV10:%.*]] = load float, ptr addrspace(4) [[MP10]], align 4
-; CHECK-NEXT: [[PROD10:%.*]] = fmul contract float [[IV10]], [[MV10]]
-; CHECK-NEXT: [[ACC10:%.*]] = fadd contract float [[ACC9]], [[PROD10]]
-; CHECK-NEXT: [[IP11:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 17
-; CHECK-NEXT: [[IV11:%.*]] = load float, ptr addrspace(1) [[IP11]], align 4
-; CHECK-NEXT: [[MP11:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 11
-; CHECK-NEXT: [[MV11:%.*]] = load float, ptr addrspace(4) [[MP11]], align 4
-; CHECK-NEXT: [[PROD11:%.*]] = fmul contract float [[IV11]], [[MV11]]
-; CHECK-NEXT: [[ACC11:%.*]] = fadd contract float [[ACC10]], [[PROD11]]
; CHECK-NEXT: [[IP12:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 18
-; CHECK-NEXT: [[IV12:%.*]] = load float, ptr addrspace(1) [[IP12]], align 4
-; CHECK-NEXT: [[MP12:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 12
-; CHECK-NEXT: [[MV12:%.*]] = load float, ptr addrspace(4) [[MP12]], align 4
-; CHECK-NEXT: [[PROD12:%.*]] = fmul contract float [[IV12]], [[MV12]]
-; CHECK-NEXT: [[ACC12:%.*]] = fadd contract float [[ACC11]], [[PROD12]]
-; CHECK-NEXT: [[IP13:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 19
-; CHECK-NEXT: [[MP13:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 13
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x float>, ptr addrspace(1) [[IP13]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = load <2 x float>, ptr addrspace(4) [[MP13]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = fmul contract <2 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x float> [[TMP3]], i64 0
-; CHECK-NEXT: [[ACC13:%.*]] = fadd contract float [[ACC12]], [[TMP4]]
-; CHECK-NEXT: [[TMP5:%.*]] = extractelement <2 x float> [[TMP3]], i64 1
-; CHECK-NEXT: [[ACC25:%.*]] = fadd contract float [[ACC13]], [[TMP5]]
+; CHECK-NEXT: [[IP14:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 20
+; CHECK-NEXT: [[IV14:%.*]] = load float, ptr addrspace(1) [[IP14]], align 4
; CHECK-NEXT: [[IP15:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 24
; CHECK-NEXT: [[IV15:%.*]] = load float, ptr addrspace(1) [[IP15]], align 4
-; CHECK-NEXT: [[MP15:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 15
-; CHECK-NEXT: [[MV15:%.*]] = load float, ptr addrspace(4) [[MP15]], align 4
-; CHECK-NEXT: [[TMP37:%.*]] = fmul contract float [[IV15]], [[MV15]]
-; CHECK-NEXT: [[ACC26:%.*]] = fadd contract float [[ACC25]], [[TMP37]]
-; CHECK-NEXT: [[IP16:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 25
-; CHECK-NEXT: [[MP16:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 16
-; CHECK-NEXT: [[TMP6:%.*]] = load <2 x float>, ptr addrspace(1) [[IP16]], align 4
-; CHECK-NEXT: [[TMP7:%.*]] = load <2 x float>, ptr addrspace(4) [[MP16]], align 4
-; CHECK-NEXT: [[TMP8:%.*]] = fmul contract <2 x float> [[TMP6]], [[TMP7]]
-; CHECK-NEXT: [[TMP9:%.*]] = extractelement <2 x float> [[TMP8]], i64 0
-; CHECK-NEXT: [[ACC16:%.*]] = fadd contract float [[ACC26]], [[TMP9]]
-; CHECK-NEXT: [[TMP10:%.*]] = extractelement <2 x float> [[TMP8]], i64 1
+; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr addrspace(1) [[IP0]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = load <2 x float>, ptr addrspace(1) [[IP6]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = load <2 x float>, ptr addrspace(1) [[IP8]], align 4
+; CHECK-NEXT: [[TMP4:%.*]] = load <2 x float>, ptr addrspace(1) [[IP10]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = load <2 x float>, ptr addrspace(1) [[IP12]], align 4
+; CHECK-NEXT: [[TMP6:%.*]] = load <16 x float>, ptr addrspace(4) [[MP0]], align 4
+; CHECK-NEXT: [[TMP7:%.*]] = insertelement <16 x float> poison, float [[IV4]], i64 4
+; CHECK-NEXT: [[TMP8:%.*]] = insertelement <16 x float> [[TMP7]], float [[IV5]], i64 5
+; CHECK-NEXT: [[TMP22:%.*]] = insertelement <16 x float> [[TMP8]], float [[IV14]], i64 14
+; CHECK-NEXT: [[TMP23:%.*]] = insertelement <16 x float> [[TMP22]], float [[IV15]], i64 15
+; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <16 x float> [[TMP23]], <16 x float> [[TMP11]], <16 x i32> <i32 16, i32 17, i32 18, i32 19, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <2 x float> [[TMP2]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP24:%.*]] = shufflevector <16 x float> [[TMP12]], <16 x float> [[TMP13]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 16, i32 17, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-NEXT: [[TMP25:%.*]] = shufflevector <2 x float> [[TMP3]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP16:%.*]] = shufflevector <16 x float> [[TMP24]], <16 x float> [[TMP25]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 16, i32 17, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-NEXT: [[TMP17:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP18:%.*]] = shufflevector <16 x float> [[TMP16]], <16 x float> [[TMP17]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 16, i32 17, i32 12, i32 13, i32 14, i32 15>
+; CHECK-NEXT: [[TMP19:%.*]] = shufflevector <2 x float> [[TMP5]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP20:%.*]] = shufflevector <16 x float> [[TMP18]], <16 x float> [[TMP19]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 16, i32 17, i32 14, i32 15>
+; CHECK-NEXT: [[TMP21:%.*]] = fmul contract <16 x float> [[TMP20]], [[TMP6]]
+; CHECK-NEXT: [[ACC14:%.*]] = extractelement <16 x float> [[TMP21]], i64 0
+; CHECK-NEXT: [[PROD15:%.*]] = extractelement <16 x float> [[TMP21]], i64 1
+; CHECK-NEXT: [[ACC15:%.*]] = fadd contract float [[ACC14]], [[PROD15]]
+; CHECK-NEXT: [[TMP9:%.*]] = extractelement <16 x float> [[TMP21]], i64 2
+; CHECK-NEXT: [[ACC16:%.*]] = fadd contract float [[ACC15]], [[TMP9]]
+; CHECK-NEXT: [[TMP10:%.*]] = extractelement <16 x float> [[TMP21]], i64 3
; CHECK-NEXT: [[ACC17:%.*]] = fadd contract float [[ACC16]], [[TMP10]]
-; CHECK-NEXT: [[IP18:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 27
-; CHECK-NEXT: [[MP18:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 18
-; CHECK-NEXT: [[TMP11:%.*]] = load <2 x float>, ptr addrspace(1) [[IP18]], align 4
-; CHECK-NEXT: [[TMP12:%.*]] = load <2 x float>, ptr addrspace(4) [[MP18]], align 4
-; CHECK-NEXT: [[TMP13:%.*]] = fmul contract <2 x float> [[TMP11]], [[TMP12]]
-; CHECK-NEXT: [[TMP14:%.*]] = extractelement <2 x float> [[TMP13]], i64 0
+; CHECK-NEXT: [[TMP14:%.*]] = extractelement <16 x float> [[TMP21]], i64 4
; CHECK-NEXT: [[ACC18:%.*]] = fadd contract float [[ACC17]], [[TMP14]]
-; CHECK-NEXT: [[TMP15:%.*]] = extractelement <2 x float> [[TMP13]], i64 1
+; CHECK-NEXT: [[TMP15:%.*]] = extractelement <16 x float> [[TMP21]], i64 5
; CHECK-NEXT: [[ACC19:%.*]] = fadd contract float [[ACC18]], [[TMP15]]
+; CHECK-NEXT: [[TMP28:%.*]] = extractelement <16 x float> [[TMP21]], i64 6
+; CHECK-NEXT: [[ACC6:%.*]] = fadd contract float [[ACC19]], [[TMP28]]
+; CHECK-NEXT: [[TMP29:%.*]] = extractelement <16 x float> [[TMP21]], i64 7
+; CHECK-NEXT: [[ACC7:%.*]] = fadd contract float [[ACC6]], [[TMP29]]
+; CHECK-NEXT: [[TMP30:%.*]] = extractelement <16 x float> [[TMP21]], i64 8
+; CHECK-NEXT: [[ACC8:%.*]] = fadd contract float [[ACC7]], [[TMP30]]
+; CHECK-NEXT: [[TMP31:%.*]] = extractelement <16 x float> [[TMP21]], i64 9
+; CHECK-NEXT: [[ACC9:%.*]] = fadd contract float [[ACC8]], [[TMP31]]
+; CHECK-NEXT: [[TMP32:%.*]] = extractelement <16 x float> [[TMP21]], i64 10
+; CHECK-NEXT: [[ACC10:%.*]] = fadd contract float [[ACC9]], [[TMP32]]
+; CHECK-NEXT: [[TMP33:%.*]] = extractelement <16 x float> [[TMP21]], i64 11
+; CHECK-NEXT: [[ACC11:%.*]] = fadd contract float [[ACC10]], [[TMP33]]
+; CHECK-NEXT: [[TMP34:%.*]] = extractelement <16 x float> [[TMP21]], i64 12
+; CHECK-NEXT: [[ACC12:%.*]] = fadd contract float [[ACC11]], [[TMP34]]
+; CHECK-NEXT: [[TMP35:%.*]] = extractelement <16 x float> [[TMP21]], i64 13
+; CHECK-NEXT: [[ACC13:%.*]] = fadd contract float [[ACC12]], [[TMP35]]
+; CHECK-NEXT: [[TMP36:%.*]] = extractelement <16 x float> [[TMP21]], i64 14
+; CHECK-NEXT: [[ACC25:%.*]] = fadd contract float [[ACC13]], [[TMP36]]
+; CHECK-NEXT: [[TMP37:%.*]] = extractelement <16 x float> [[TMP21]], i64 15
+; CHECK-NEXT: [[ACC26:%.*]] = fadd contract float [[ACC25]], [[TMP37]]
+; CHECK-NEXT: [[IP16:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 25
+; CHECK-NEXT: [[MP16:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 16
; CHECK-NEXT: [[IP20:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 32
-; CHECK-NEXT: [[IV20:%.*]] = load float, ptr addrspace(1) [[IP20]], align 4
-; CHECK-NEXT: [[MP20:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 20
+; CHECK-NEXT: [[TMP38:%.*]] = load <4 x float>, ptr addrspace(1) [[IP16]], align 4
+; CHECK-NEXT: [[TMP39:%.*]] = load <4 x float>, ptr addrspace(1) [[IP20]], align 4
+; CHECK-NEXT: [[TMP40:%.*]] = load <8 x float>, ptr addrspace(4) [[MP16]], align 4
+; CHECK-NEXT: [[TMP41:%.*]] = shufflevector <4 x float> [[TMP38]], <4 x float> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP42:%.*]] = shufflevector <4 x float> [[TMP39]], <4 x float> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP43:%.*]] = shufflevector <4 x float> [[TMP38]], <4 x float> [[TMP39]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT: [[TMP44:%.*]] = fmul contract <8 x float> [[TMP43]], [[TMP40]]
+; CHECK-NEXT: [[TMP45:%.*]] = extractelement <8 x float> [[TMP44]], i64 0
+; CHECK-NEXT: [[ACC27:%.*]] = fadd contract float [[ACC26]], [[TMP45]]
+; CHECK-NEXT: [[TMP46:%.*]] = extractelement <8 x float> [[TMP44]], i64 1
+; CHECK-NEXT: [[ACC28:%.*]] = fadd contract float [[ACC27]], [[TMP46]]
+; CHECK-NEXT: [[TMP47:%.*]] = extractelement <8 x float> [[TMP44]], i64 2
+; CHECK-NEXT: [[ACC29:%.*]] = fadd contract float [[ACC28]], [[TMP47]]
+; CHECK-NEXT: [[TMP48:%.*]] = extractelement <8 x float> [[TMP44]], i64 3
+; CHECK-NEXT: [[ACC30:%.*]] = fadd contract float [[ACC29]], [[TMP48]]
+; CHECK-NEXT: [[TMP49:%.*]] = extractelement <8 x float> [[TMP44]], i64 4
+; CHECK-NEXT: [[ACC20:%.*]] = fadd contract float [[ACC30]], [[TMP49]]
+; CHECK-NEXT: [[TMP50:%.*]] = extractelement <8 x float> [[TMP44]], i64 5
+; CHECK-NEXT: [[ACC21:%.*]] = fadd contract float [[ACC20]], [[TMP50]]
+; CHECK-NEXT: [[TMP51:%.*]] = extractelement <8 x float> [[TMP44]], i64 6
+; CHECK-NEXT: [[ACC22:%.*]] = fadd contract float [[ACC21]], [[TMP51]]
+; CHECK-NEXT: [[TMP52:%.*]] = extractelement <8 x float> [[TMP44]], i64 7
+; CHECK-NEXT: [[ACC23:%.*]] = fadd contract float [[ACC22]], [[TMP52]]
+; CHECK-NEXT: [[IP24:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 36
+; CHECK-NEXT: [[IV20:%.*]] = load float, ptr addrspace(1) [[IP24]], align 4
+; CHECK-NEXT: [[MP20:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 24
; CHECK-NEXT: [[MV20:%.*]] = load float, ptr addrspace(4) [[MP20]], align 4
; CHECK-NEXT: [[PROD20:%.*]] = fmul contract float [[IV20]], [[MV20]]
-; CHECK-NEXT: [[ACC20:%.*]] = fadd contract float [[ACC19]], [[PROD20]]
-; CHECK-NEXT: [[IP21:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 33
-; CHECK-NEXT: [[MP21:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 21
-; CHECK-NEXT: [[TMP16:%.*]] = load <2 x float>, ptr addrspace(1) [[IP21]], align 4
-; CHECK-NEXT: [[TMP17:%.*]] = load <2 x float>, ptr addrspace(4) [[MP21]], align 4
-; CHECK-NEXT: [[TMP18:%.*]] = fmul contract <2 x float> [[TMP16]], [[TMP17]]
-; CHECK-NEXT: [[TMP19:%.*]] = extractelement <2 x float> [[TMP18]], i64 0
-; CHECK-NEXT: [[ACC21:%.*]] = fadd contract float [[ACC20]], [[TMP19]]
-; CHECK-NEXT: [[TMP20:%.*]] = extractelement <2 x float> [[TMP18]], i64 1
-; CHECK-NEXT: [[ACC22:%.*]] = fadd contract float [[ACC21]], [[TMP20]]
-; CHECK-NEXT: [[IP23:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 35
-; CHECK-NEXT: [[MP23:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 23
-; CHECK-NEXT: [[TMP21:%.*]] = load <2 x float>, ptr addrspace(1) [[IP23]], align 4
-; CHECK-NEXT: [[TMP22:%.*]] = load <2 x float>, ptr addrspace(4) [[MP23]], align 4
-; CHECK-NEXT: [[TMP23:%.*]] = fmul contract <2 x float> [[TMP21]], [[TMP22]]
-; CHECK-NEXT: [[TMP24:%.*]] = extractelement <2 x float> [[TMP23]], i64 0
-; CHECK-NEXT: [[ACC23:%.*]] = fadd contract float [[ACC22]], [[TMP24]]
-; CHECK-NEXT: [[TMP25:%.*]] = extractelement <2 x float> [[TMP23]], i64 1
-; CHECK-NEXT: [[ACC24:%.*]] = fadd contract float [[ACC23]], [[TMP25]]
+; CHECK-NEXT: [[ACC24:%.*]] = fadd contract float [[ACC23]], [[PROD20]]
; CHECK-NEXT: ret float [[ACC24]]
;
%ip0 = getelementptr inbounds float, ptr addrspace(1) %input, i64 0
@@ -299,139 +263,103 @@ define float @conv_contract(ptr addrspace(1) %input, ptr addrspace(4) %mask) {
define float @conv_no_contract(ptr addrspace(1) %input, ptr addrspace(4) %mask) {
; CHECK-LABEL: @conv_no_contract(
; CHECK-NEXT: [[IP0:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT:%.*]], i64 0
-; CHECK-NEXT: [[IV0:%.*]] = load float, ptr addrspace(1) [[IP0]], align 4
; CHECK-NEXT: [[MP0:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK:%.*]], i64 0
-; CHECK-NEXT: [[MV0:%.*]] = load float, ptr addrspace(4) [[MP0]], align 4
-; CHECK-NEXT: [[PROD0:%.*]] = fmul float [[IV0]], [[MV0]]
-; CHECK-NEXT: [[IP1:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 1
-; CHECK-NEXT: [[IV1:%.*]] = load float, ptr addrspace(1) [[IP1]], align 4
-; CHECK-NEXT: [[MP1:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 1
-; CHECK-NEXT: [[MV1:%.*]] = load float, ptr addrspace(4) [[MP1]], align 4
-; CHECK-NEXT: [[PROD1:%.*]] = fmul float [[IV1]], [[MV1]]
-; CHECK-NEXT: [[ACC1:%.*]] = fadd float [[PROD0]], [[PROD1]]
-; CHECK-NEXT: [[IP2:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 2
-; CHECK-NEXT: [[IV2:%.*]] = load float, ptr addrspace(1) [[IP2]], align 4
-; CHECK-NEXT: [[MP2:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 2
-; CHECK-NEXT: [[MV2:%.*]] = load float, ptr addrspace(4) [[MP2]], align 4
-; CHECK-NEXT: [[PROD2:%.*]] = fmul float [[IV2]], [[MV2]]
-; CHECK-NEXT: [[ACC2:%.*]] = fadd float [[ACC1]], [[PROD2]]
-; CHECK-NEXT: [[IP3:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 3
-; CHECK-NEXT: [[IV3:%.*]] = load float, ptr addrspace(1) [[IP3]], align 4
-; CHECK-NEXT: [[MP3:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 3
-; CHECK-NEXT: [[MV3:%.*]] = load float, ptr addrspace(4) [[MP3]], align 4
-; CHECK-NEXT: [[PROD3:%.*]] = fmul float [[IV3]], [[MV3]]
-; CHECK-NEXT: [[ACC3:%.*]] = fadd float [[ACC2]], [[PROD3]]
; CHECK-NEXT: [[IP4:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 4
; CHECK-NEXT: [[IV4:%.*]] = load float, ptr addrspace(1) [[IP4]], align 4
-; CHECK-NEXT: [[MP4:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 4
-; CHECK-NEXT: [[MV4:%.*]] = load float, ptr addrspace(4) [[MP4]], align 4
-; CHECK-NEXT: [[PROD4:%.*]] = fmul float [[IV4]], [[MV4]]
-; CHECK-NEXT: [[ACC4:%.*]] = fadd float [[ACC3]], [[PROD4]]
; CHECK-NEXT: [[IP5:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 8
; CHECK-NEXT: [[IV5:%.*]] = load float, ptr addrspace(1) [[IP5]], align 4
-; CHECK-NEXT: [[MP5:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 5
-; CHECK-NEXT: [[MV5:%.*]] = load float, ptr addrspace(4) [[MP5]], align 4
-; CHECK-NEXT: [[PROD5:%.*]] = fmul float [[IV5]], [[MV5]]
-; CHECK-NEXT: [[ACC5:%.*]] = fadd float [[ACC4]], [[PROD5]]
; CHECK-NEXT: [[IP6:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 9
-; CHECK-NEXT: [[IV6:%.*]] = load float, ptr addrspace(1) [[IP6]], align 4
-; CHECK-NEXT: [[MP6:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 6
-; CHECK-NEXT: [[MV6:%.*]] = load float, ptr addrspace(4) [[MP6]], align 4
-; CHECK-NEXT: [[PROD6:%.*]] = fmul float [[IV6]], [[MV6]]
-; CHECK-NEXT: [[ACC6:%.*]] = fadd float [[ACC5]], [[PROD6]]
-; CHECK-NEXT: [[IP7:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 10
-; CHECK-NEXT: [[IV7:%.*]] = load float, ptr addrspace(1) [[IP7]], align 4
-; CHECK-NEXT: [[MP7:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 7
-; CHECK-NEXT: [[MV7:%.*]] = load float, ptr addrspace(4) [[MP7]], align 4
-; CHECK-NEXT: [[PROD7:%.*]] = fmul float [[IV7]], [[MV7]]
-; CHECK-NEXT: [[ACC7:%.*]] = fadd float [[ACC6]], [[PROD7]]
; CHECK-NEXT: [[IP8:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 11
-; CHECK-NEXT: [[IV8:%.*]] = load float, ptr addrspace(1) [[IP8]], align 4
-; CHECK-NEXT: [[MP8:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 8
-; CHECK-NEXT: [[MV8:%.*]] = load float, ptr addrspace(4) [[MP8]], align 4
-; CHECK-NEXT: [[PROD8:%.*]] = fmul float [[IV8]], [[MV8]]
-; CHECK-NEXT: [[ACC8:%.*]] = fadd float [[ACC7]], [[PROD8]]
-; CHECK-NEXT: [[IP9:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 12
-; CHECK-NEXT: [[IV9:%.*]] = load float, ptr addrspace(1) [[IP9]], align 4
-; CHECK-NEXT: [[MP9:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 9
-; CHECK-NEXT: [[MV9:%.*]] = load float, ptr addrspace(4) [[MP9]], align 4
-; CHECK-NEXT: [[PROD9:%.*]] = fmul float [[IV9]], [[MV9]]
-; CHECK-NEXT: [[ACC9:%.*]] = fadd float [[ACC8]], [[PROD9]]
; CHECK-NEXT: [[IP10:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 16
-; CHECK-NEXT: [[IV10:%.*]] = load float, ptr addrspace(1) [[IP10]], align 4
-; CHECK-NEXT: [[MP10:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 10
-; CHECK-NEXT: [[MV10:%.*]] = load float, ptr addrspace(4) [[MP10]], align 4
-; CHECK-NEXT: [[PROD10:%.*]] = fmul float [[IV10]], [[MV10]]
-; CHECK-NEXT: [[ACC10:%.*]] = fadd float [[ACC9]], [[PROD10]]
-; CHECK-NEXT: [[IP11:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 17
-; CHECK-NEXT: [[IV11:%.*]] = load float, ptr addrspace(1) [[IP11]], align 4
-; CHECK-NEXT: [[MP11:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 11
-; CHECK-NEXT: [[MV11:%.*]] = load float, ptr addrspace(4) [[MP11]], align 4
-; CHECK-NEXT: [[PROD11:%.*]] = fmul float [[IV11]], [[MV11]]
-; CHECK-NEXT: [[ACC11:%.*]] = fadd float [[ACC10]], [[PROD11]]
; CHECK-NEXT: [[IP12:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 18
-; CHECK-NEXT: [[IV12:%.*]] = load float, ptr addrspace(1) [[IP12]], align 4
-; CHECK-NEXT: [[MP12:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 12
-; CHECK-NEXT: [[MV12:%.*]] = load float, ptr addrspace(4) [[MP12]], align 4
-; CHECK-NEXT: [[PROD12:%.*]] = fmul float [[IV12]], [[MV12]]
-; CHECK-NEXT: [[ACC12:%.*]] = fadd float [[ACC11]], [[PROD12]]
-; CHECK-NEXT: [[IP13:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 19
-; CHECK-NEXT: [[MP13:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 13
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x float>, ptr addrspace(1) [[IP13]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = load <2 x float>, ptr addrspace(4) [[MP13]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = fmul <2 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x float> [[TMP3]], i64 0
-; CHECK-NEXT: [[ACC13:%.*]] = fadd float [[ACC12]], [[TMP4]]
-; CHECK-NEXT: [[TMP5:%.*]] = extractelement <2 x float> [[TMP3]], i64 1
-; CHECK-NEXT: [[ACC14:%.*]] = fadd float [[ACC13]], [[TMP5]]
+; CHECK-NEXT: [[IP14:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 20
+; CHECK-NEXT: [[IV14:%.*]] = load float, ptr addrspace(1) [[IP14]], align 4
; CHECK-NEXT: [[IP15:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 24
; CHECK-NEXT: [[IV15:%.*]] = load float, ptr addrspace(1) [[IP15]], align 4
-; CHECK-NEXT: [[MP15:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 15
-; CHECK-NEXT: [[MV15:%.*]] = load float, ptr addrspace(4) [[MP15]], align 4
-; CHECK-NEXT: [[TMP37:%.*]] = fmul float [[IV15]], [[MV15]]
+; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr addrspace(1) [[IP0]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = load <2 x float>, ptr addrspace(1) [[IP6]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = load <2 x float>, ptr addrspace(1) [[IP8]], align 4
+; CHECK-NEXT: [[TMP4:%.*]] = load <2 x float>, ptr addrspace(1) [[IP10]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = load <2 x float>, ptr addrspace(1) [[IP12]], align 4
+; CHECK-NEXT: [[TMP6:%.*]] = load <16 x float>, ptr addrspace(4) [[MP0]], align 4
+; CHECK-NEXT: [[TMP7:%.*]] = insertelement <16 x float> poison, float [[IV4]], i64 4
+; CHECK-NEXT: [[TMP8:%.*]] = insertelement <16 x float> [[TMP7]], float [[IV5]], i64 5
+; CHECK-NEXT: [[TMP9:%.*]] = insertelement <16 x float> [[TMP8]], float [[IV14]], i64 14
+; CHECK-NEXT: [[TMP10:%.*]] = insertelement <16 x float> [[TMP9]], float [[IV15]], i64 15
+; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <16 x float> [[TMP10]], <16 x float> [[TMP11]], <16 x i32> <i32 16, i32 17, i32 18, i32 19, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <2 x float> [[TMP2]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP14:%.*]] = shufflevector <16 x float> [[TMP12]], <16 x float> [[TMP13]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 16, i32 17, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-NEXT: [[TMP15:%.*]] = shufflevector <2 x float> [[TMP3]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP16:%.*]] = shufflevector <16 x float> [[TMP14]], <16 x float> [[TMP15]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 16, i32 17, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-NEXT: [[TMP17:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP18:%.*]] = shufflevector <16 x float> [[TMP16]], <16 x float> [[TMP17]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 16, i32 17, i32 12, i32 13, i32 14, i32 15>
+; CHECK-NEXT: [[TMP19:%.*]] = shufflevector <2 x float> [[TMP5]], <2 x float> poison, <16 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP20:%.*]] = shufflevector <16 x float> [[TMP18]], <16 x float> [[TMP19]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 16, i32 17, i32 14, i32 15>
+; CHECK-NEXT: [[TMP21:%.*]] = fmul <16 x float> [[TMP20]], [[TMP6]]
+; CHECK-NEXT: [[TMP22:%.*]] = extractelement <16 x float> [[TMP21]], i64 0
+; CHECK-NEXT: [[TMP23:%.*]] = extractelement <16 x float> [[TMP21]], i64 1
+; CHECK-NEXT: [[ACC1:%.*]] = fadd float [[TMP22]], [[TMP23]]
+; CHECK-NEXT: [[TMP24:%.*]] = extractelement <16 x float> [[TMP21]], i64 2
+; CHECK-NEXT: [[ACC2:%.*]] = fadd float [[ACC1]], [[TMP24]]
+; CHECK-NEXT: [[TMP25:%.*]] = extractelement <16 x float> [[TMP21]], i64 3
+; CHECK-NEXT: [[ACC3:%.*]] = fadd float [[ACC2]], [[TMP25]]
+; CHECK-NEXT: [[TMP26:%.*]] = extractelement <16 x float> [[TMP21]], i64 4
+; CHECK-NEXT: [[ACC4:%.*]] = fadd float [[ACC3]], [[TMP26]]
+; CHECK-NEXT: [[TMP27:%.*]] = extractelement <16 x float> [[TMP21]], i64 5
+; CHECK-NEXT: [[ACC5:%.*]] = fadd float [[ACC4]], [[TMP27]]
+; CHECK-NEXT: [[TMP28:%.*]] = extractelement <16 x float> [[TMP21]], i64 6
+; CHECK-NEXT: [[ACC6:%.*]] = fadd float [[ACC5]], [[TMP28]]
+; CHECK-NEXT: [[TMP29:%.*]] = extractelement <16 x float> [[TMP21]], i64 7
+; CHECK-NEXT: [[ACC7:%.*]] = fadd float [[ACC6]], [[TMP29]]
+; CHECK-NEXT: [[TMP30:%.*]] = extractelement <16 x float> [[TMP21]], i64 8
+; CHECK-NEXT: [[ACC8:%.*]] = fadd float [[ACC7]], [[TMP30]]
+; CHECK-NEXT: [[TMP31:%.*]] = extractelement <16 x float> [[TMP21]], i64 9
+; CHECK-NEXT: [[ACC9:%.*]] = fadd float [[ACC8]], [[TMP31]]
+; CHECK-NEXT: [[TMP32:%.*]] = extractelement <16 x float> [[TMP21]], i64 10
+; CHECK-NEXT: [[ACC10:%.*]] = fadd float [[ACC9]], [[TMP32]]
+; CHECK-NEXT: [[TMP33:%.*]] = extractelement <16 x float> [[TMP21]], i64 11
+; CHECK-NEXT: [[ACC11:%.*]] = fadd float [[ACC10]], [[TMP33]]
+; CHECK-NEXT: [[TMP34:%.*]] = extractelement <16 x float> [[TMP21]], i64 12
+; CHECK-NEXT: [[ACC12:%.*]] = fadd float [[ACC11]], [[TMP34]]
+; CHECK-NEXT: [[TMP35:%.*]] = extractelement <16 x float> [[TMP21]], i64 13
+; CHECK-NEXT: [[ACC13:%.*]] = fadd float [[ACC12]], [[TMP35]]
+; CHECK-NEXT: [[TMP36:%.*]] = extractelement <16 x float> [[TMP21]], i64 14
+; CHECK-NEXT: [[ACC14:%.*]] = fadd float [[ACC13]], [[TMP36]]
+; CHECK-NEXT: [[TMP37:%.*]] = extractelement <16 x float> [[TMP21]], i64 15
; CHECK-NEXT: [[ACC15:%.*]] = fadd float [[ACC14]], [[TMP37]]
; CHECK-NEXT: [[IP16:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 25
; CHECK-NEXT: [[MP16:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 16
-; CHECK-NEXT: [[TMP6:%.*]] = load <2 x float>, ptr addrspace(1) [[IP16]], align 4
-; CHECK-NEXT: [[TMP7:%.*]] = load <2 x float>, ptr addrspace(4) [[MP16]], align 4
-; CHECK-NEXT: [[TMP8:%.*]] = fmul <2 x float> [[TMP6]], [[TMP7]]
-; CHECK-NEXT: [[TMP9:%.*]] = extractelement <2 x float> [[TMP8]], i64 0
-; CHECK-NEXT: [[ACC16:%.*]] = fadd float [[ACC15]], [[TMP9]]
-; CHECK-NEXT: [[TMP10:%.*]] = extractelement <2 x float> [[TMP8]], i64 1
-; CHECK-NEXT: [[ACC17:%.*]] = fadd float [[ACC16]], [[TMP10]]
-; CHECK-NEXT: [[IP18:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 27
-; CHECK-NEXT: [[MP18:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 18
-; CHECK-NEXT: [[TMP11:%.*]] = load <2 x float>, ptr addrspace(1) [[IP18]], align 4
-; CHECK-NEXT: [[TMP12:%.*]] = load <2 x float>, ptr addrspace(4) [[MP18]], align 4
-; CHECK-NEXT: [[TMP13:%.*]] = fmul <2 x float> [[TMP11]], [[TMP12]]
-; CHECK-NEXT: [[TMP14:%.*]] = extractelement <2 x float> [[TMP13]], i64 0
-; CHECK-NEXT: [[ACC18:%.*]] = fadd float [[ACC17]], [[TMP14]]
-; CHECK-NEXT: [[TMP15:%.*]] = extractelement <2 x float> [[TMP13]], i64 1
-; CHECK-NEXT: [[ACC19:%.*]] = fadd float [[ACC18]], [[TMP15]]
; CHECK-NEXT: [[IP20:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 32
-; CHECK-NEXT: [[IV24:%.*]] = load float, ptr addrspace(1) [[IP20]], align 4
-; CHECK-NEXT: [[MP24:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 20
+; CHECK-NEXT: [[TMP38:%.*]] = load <4 x float>, ptr addrspace(1) [[IP16]], align 4
+; CHECK-NEXT: [[TMP39:%.*]] = load <4 x float>, ptr addrspace(1) [[IP20]], align 4
+; CHECK-NEXT: [[TMP40:%.*]] = load <8 x float>, ptr addrspace(4) [[MP16]], align 4
+; CHECK-NEXT: [[TMP41:%.*]] = shufflevector <4 x float> [[TMP38]], <4 x float> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP42:%.*]] = shufflevector <4 x float> [[TMP39]], <4 x float> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP43:%.*]] = shufflevector <4 x float> [[TMP38]], <4 x float> [[TMP39]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT: [[TMP44:%.*]] = fmul <8 x float> [[TMP43]], [[TMP40]]
+; CHECK-NEXT: [[TMP45:%.*]] = extractelement <8 x float> [[TMP44]], i64 0
+; CHECK-NEXT: [[ACC16:%.*]] = fadd float [[ACC15]], [[TMP45]]
+; CHECK-NEXT: [[TMP46:%.*]] = extractelement <8 x float> [[TMP44]], i64 1
+; CHECK-NEXT: [[ACC17:%.*]] = fadd float [[ACC16]], [[TMP46]]
+; CHECK-NEXT: [[TMP47:%.*]] = extractelement <8 x float> [[TMP44]], i64 2
+; CHECK-NEXT: [[ACC18:%.*]] = fadd float [[ACC17]], [[TMP47]]
+; CHECK-NEXT: [[TMP48:%.*]] = extractelement <8 x float> [[TMP44]], i64 3
+; CHECK-NEXT: [[ACC19:%.*]] = fadd float [[ACC18]], [[TMP48]]
+; CHECK-NEXT: [[TMP49:%.*]] = extractelement <8 x float> [[TMP44]], i64 4
+; CHECK-NEXT: [[ACC20:%.*]] = fadd float [[ACC19]], [[TMP49]]
+; CHECK-NEXT: [[TMP50:%.*]] = extractelement <8 x float> [[TMP44]], i64 5
+; CHECK-NEXT: [[ACC21:%.*]] = fadd float [[ACC20]], [[TMP50]]
+; CHECK-NEXT: [[TMP51:%.*]] = extractelement <8 x float> [[TMP44]], i64 6
+; CHECK-NEXT: [[ACC22:%.*]] = fadd float [[ACC21]], [[TMP51]]
+; CHECK-NEXT: [[TMP52:%.*]] = extractelement <8 x float> [[TMP44]], i64 7
+; CHECK-NEXT: [[ACC23:%.*]] = fadd float [[ACC22]], [[TMP52]]
+; CHECK-NEXT: [[IP24:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 36
+; CHECK-NEXT: [[IV24:%.*]] = load float, ptr addrspace(1) [[IP24]], align 4
+; CHECK-NEXT: [[MP24:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 24
; CHECK-NEXT: [[MV24:%.*]] = load float, ptr addrspace(4) [[MP24]], align 4
; CHECK-NEXT: [[PROD24:%.*]] = fmul float [[IV24]], [[MV24]]
-; CHECK-NEXT: [[ACC20:%.*]] = fadd float [[ACC19]], [[PROD24]]
-; CHECK-NEXT: [[IP21:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 33
-; CHECK-NEXT: [[MP21:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 21
-; CHECK-NEXT: [[TMP16:%.*]] = load <2 x float>, ptr addrspace(1) [[IP21]], align 4
-; CHECK-NEXT: [[TMP17:%.*]] = load <2 x float>, ptr addrspace(4) [[MP21]], align 4
-; CHECK-NEXT: [[TMP18:%.*]] = fmul <2 x float> [[TMP16]], [[TMP17]]
-; CHECK-NEXT: [[TMP19:%.*]] = extractelement <2 x float> [[TMP18]], i64 0
-; CHECK-NEXT: [[ACC21:%.*]] = fadd float [[ACC20]], [[TMP19]]
-; CHECK-NEXT: [[TMP20:%.*]] = extractelement <2 x float> [[TMP18]], i64 1
-; CHECK-NEXT: [[ACC22:%.*]] = fadd float [[ACC21]], [[TMP20]]
-; CHECK-NEXT: [[IP23:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[INPUT]], i64 35
-; CHECK-NEXT: [[MP23:%.*]] = getelementptr inbounds float, ptr addrspace(4) [[MASK]], i64 23
-; CHECK-NEXT: [[TMP21:%.*]] = load <2 x float>, ptr addrspace(1) [[IP23]], align 4
-; CHECK-NEXT: [[TMP22:%.*]] = load <2 x float>, ptr addrspace(4) [[MP23]], align 4
-; CHECK-NEXT: [[TMP23:%.*]] = fmul <2 x float> [[TMP21]], [[TMP22]]
-; CHECK-NEXT: [[TMP24:%.*]] = extractelement <2 x float> [[TMP23]], i64 0
-; CHECK-NEXT: [[ACC23:%.*]] = fadd float [[ACC22]], [[TMP24]]
-; CHECK-NEXT: [[TMP25:%.*]] = extractelement <2 x float> [[TMP23]], i64 1
-; CHECK-NEXT: [[ACC24:%.*]] = fadd float [[ACC23]], [[TMP25]]
+; CHECK-NEXT: [[ACC24:%.*]] = fadd float [[ACC23]], [[PROD24]]
; CHECK-NEXT: ret float [[ACC24]]
;
%ip0 = getelementptr inbounds float, ptr addrspace(1) %input, i64 0
diff --git a/llvm/test/Transforms/VectorCombine/AMDGPU/combine-scalar-selects.ll b/llvm/test/Transforms/VectorCombine/AMDGPU/combine-scalar-selects.ll
index 97ee47877ec2d..78b58dc3e02fb 100644
--- a/llvm/test/Transforms/VectorCombine/AMDGPU/combine-scalar-selects.ll
+++ b/llvm/test/Transforms/VectorCombine/AMDGPU/combine-scalar-selects.ll
@@ -1024,38 +1024,31 @@ define amdgpu_kernel void @combine_v4f32_to_v8i16(
; CHECK-OPT-LABEL: define amdgpu_kernel void @combine_v4f32_to_v8i16(
; CHECK-OPT-SAME: ptr addrspace(1) [[OUT:%.*]], <4 x float> [[SRC:%.*]], i1 [[COND:%.*]]) {
; CHECK-OPT-NEXT: [[ENTRY:.*:]]
-; CHECK-OPT-NEXT: [[HALVES:%.*]] = bitcast <4 x float> [[SRC]] to <8 x i16>
-; CHECK-OPT-NEXT: [[E0:%.*]] = extractelement <8 x i16> [[HALVES]], i64 0
-; CHECK-OPT-NEXT: [[E1:%.*]] = extractelement <8 x i16> [[HALVES]], i64 1
-; CHECK-OPT-NEXT: [[E2:%.*]] = extractelement <8 x i16> [[HALVES]], i64 2
-; CHECK-OPT-NEXT: [[E3:%.*]] = extractelement <8 x i16> [[HALVES]], i64 3
-; CHECK-OPT-NEXT: [[E4:%.*]] = extractelement <8 x i16> [[HALVES]], i64 4
-; CHECK-OPT-NEXT: [[E5:%.*]] = extractelement <8 x i16> [[HALVES]], i64 5
-; CHECK-OPT-NEXT: [[E6:%.*]] = extractelement <8 x i16> [[HALVES]], i64 6
-; CHECK-OPT-NEXT: [[E7:%.*]] = extractelement <8 x i16> [[HALVES]], i64 7
-; CHECK-OPT-NEXT: [[S0:%.*]] = select i1 [[COND]], i16 [[E0]], i16 0
-; CHECK-OPT-NEXT: [[S1:%.*]] = select i1 [[COND]], i16 [[E1]], i16 0
-; CHECK-OPT-NEXT: [[S2:%.*]] = select i1 [[COND]], i16 [[E2]], i16 0
-; CHECK-OPT-NEXT: [[S3:%.*]] = select i1 [[COND]], i16 [[E3]], i16 0
-; CHECK-OPT-NEXT: [[S4:%.*]] = select i1 [[COND]], i16 [[E4]], i16 0
-; CHECK-OPT-NEXT: [[S5:%.*]] = select i1 [[COND]], i16 [[E5]], i16 0
-; CHECK-OPT-NEXT: [[S6:%.*]] = select i1 [[COND]], i16 [[E6]], i16 0
-; CHECK-OPT-NEXT: [[S7:%.*]] = select i1 [[COND]], i16 [[E7]], i16 0
-; CHECK-OPT-NEXT: store i16 [[S0]], ptr addrspace(1) [[OUT]], align 2
+; CHECK-OPT-NEXT: [[COMBINED_SEL:%.*]] = select i1 [[COND]], <4 x float> [[SRC]], <4 x float> zeroinitializer
+; CHECK-OPT-NEXT: [[COMBINED_BC:%.*]] = bitcast <4 x float> [[COMBINED_SEL]] to <8 x i16>
+; CHECK-OPT-NEXT: [[TMP0:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 0
+; CHECK-OPT-NEXT: [[TMP3:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 1
+; CHECK-OPT-NEXT: [[TMP5:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 2
+; CHECK-OPT-NEXT: [[TMP7:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 3
+; CHECK-OPT-NEXT: [[TMP2:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 4
+; CHECK-OPT-NEXT: [[TMP4:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 5
+; CHECK-OPT-NEXT: [[TMP6:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 6
+; CHECK-OPT-NEXT: [[TMP1:%.*]] = extractelement <8 x i16> [[COMBINED_BC]], i64 7
+; CHECK-OPT-NEXT: store i16 [[TMP0]], ptr addrspace(1) [[OUT]], align 2
; CHECK-OPT-NEXT: [[PTR1:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 1
-; CHECK-OPT-NEXT: store i16 [[S1]], ptr addrspace(1) [[PTR1]], align 2
+; CHECK-OPT-NEXT: store i16 [[TMP3]], ptr addrspace(1) [[PTR1]], align 2
; CHECK-OPT-NEXT: [[PTR2:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 2
-; CHECK-OPT-NEXT: store i16 [[S2]], ptr addrspace(1) [[PTR2]], align 2
+; CHECK-OPT-NEXT: store i16 [[TMP5]], ptr addrspace(1) [[PTR2]], align 2
; CHECK-OPT-NEXT: [[PTR3:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 3
-; CHECK-OPT-NEXT: store i16 [[S3]], ptr addrspace(1) [[PTR3]], align 2
+; CHECK-OPT-NEXT: store i16 [[TMP7]], ptr addrspace(1) [[PTR3]], align 2
; CHECK-OPT-NEXT: [[PTR4:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 4
-; CHECK-OPT-NEXT: store i16 [[S4]], ptr addrspace(1) [[PTR4]], align 2
+; CHECK-OPT-NEXT: store i16 [[TMP2]], ptr addrspace(1) [[PTR4]], align 2
; CHECK-OPT-NEXT: [[PTR5:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 5
-; CHECK-OPT-NEXT: store i16 [[S5]], ptr addrspace(1) [[PTR5]], align 2
+; CHECK-OPT-NEXT: store i16 [[TMP4]], ptr addrspace(1) [[PTR5]], align 2
; CHECK-OPT-NEXT: [[PTR6:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 6
-; CHECK-OPT-NEXT: store i16 [[S6]], ptr addrspace(1) [[PTR6]], align 2
+; CHECK-OPT-NEXT: store i16 [[TMP6]], ptr addrspace(1) [[PTR6]], align 2
; CHECK-OPT-NEXT: [[PTR7:%.*]] = getelementptr i16, ptr addrspace(1) [[OUT]], i64 7
-; CHECK-OPT-NEXT: store i16 [[S7]], ptr addrspace(1) [[PTR7]], align 2
+; CHECK-OPT-NEXT: store i16 [[TMP1]], ptr addrspace(1) [[PTR7]], align 2
; CHECK-OPT-NEXT: ret void
;
; CHECK-NOOPT-LABEL: define amdgpu_kernel void @combine_v4f32_to_v8i16(
>From de3633254cc16045ff91a5e920fe2ba29437e39e Mon Sep 17 00:00:00 2001
From: Akash Dutta <Akash.Dutta at amd.com>
Date: Mon, 21 Sep 2026 17:46:24 +0000
Subject: [PATCH 11/11] revert redundant test changes
---
.../notriviallyvectorizableintrinsicoperands.ll | 12 ++++++------
1 file changed, 6 insertions(+), 6 deletions(-)
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
index b2b5ac09a2426..533f8bce9ed18 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/notriviallyvectorizableintrinsicoperands.ll
@@ -543,8 +543,8 @@ define amdgpu_kernel void @test_single_exp_hreduction(
; GCN-NEXT: [[ENTRY:.*:]]
; GCN-NEXT: [[P0:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 0
; GCN-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr addrspace(1) [[P0]], align 4
-; GCN-NEXT: [[SUM:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP0]])
-; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[SUM]])
+; GCN-NEXT: [[TMP1:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP0]])
+; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP1]])
; GCN-NEXT: store float [[EXP0]], ptr addrspace(1) [[OUTPUT]], align 4
; GCN-NEXT: ret void
;
@@ -577,10 +577,10 @@ define amdgpu_kernel void @test_hreduction_into_exp(
; GCN-NEXT: [[P4:%.*]] = getelementptr float, ptr addrspace(1) [[INPUT]], i64 4
; GCN-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr addrspace(1) [[P0]], align 4
; GCN-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr addrspace(1) [[P4]], align 4
-; GCN-NEXT: [[TMP8:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP0]])
-; GCN-NEXT: [[TMP9:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP1]])
-; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP8]])
-; GCN-NEXT: [[EXP1:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP9]])
+; GCN-NEXT: [[TMP2:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP0]])
+; GCN-NEXT: [[TMP3:%.*]] = call fast float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[TMP1]])
+; GCN-NEXT: [[EXP0:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP2]])
+; GCN-NEXT: [[EXP1:%.*]] = tail call float @llvm.amdgcn.exp2.f32(float [[TMP3]])
; GCN-NEXT: [[VEC0:%.*]] = insertelement <2 x float> poison, float [[EXP0]], i64 0
; GCN-NEXT: [[VEC1:%.*]] = insertelement <2 x float> [[VEC0]], float [[EXP1]], i64 1
; GCN-NEXT: [[VEC_I32:%.*]] = bitcast <2 x float> [[VEC1]] to <2 x i32>
More information about the llvm-commits
mailing list