[llvm] [AMDGPU] Add getCmpSelInstrCost override to re-enable SimplifyCFG speculation for vector types (PR #208043)
Frederik Harwath via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 28 05:58:13 PDT 2026
https://github.com/frederik-h updated https://github.com/llvm/llvm-project/pull/208043
>From 9d67adcc414ed2641efab98c72b785eeeccbaa99 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Mon, 15 Jun 2026 10:38:37 -0400
Subject: [PATCH 01/20] [SimplifyCFG] Document PHI folding threshold behavior
Starting with a generic cost model change in commit 0967957d7a94
"[CostModel] Handle all cost kinds in getCmpSelInstrCost (#148233)",
SimplifyCFG has stopped folding branches that really should be folded
on AMDGPU since not folding them causes significant spilling.
Add a test demonstrating the spilling and a simple test demonstrating
the missed fold due to the cost change.
---
.../AMDGPU/speculate-block-regpressure.ll | 197 ++++++++++++++++++
.../SimplifyCFG/AMDGPU/speculate-block.ll | 44 ++++
2 files changed, 241 insertions(+)
create mode 100644 llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
create mode 100644 llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll
diff --git a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
new file mode 100644
index 0000000000000..91914bec8cdb0
--- /dev/null
+++ b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
@@ -0,0 +1,197 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a < %s | FileCheck %s --check-prefix=IR
+; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a < %s \
+; RUN: | llc -O3 -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a | FileCheck %s --check-prefix=ASM
+; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -phi-node-folding-threshold=4 < %s \
+; RUN: | llc -O3 -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a | FileCheck %s --check-prefix=ASM-THRESHOLD4
+
+; Regression test for SGPR spilling caused by commit 0967957d7a94
+; "[CostModel] Handle all cost kinds in getCmpSelInstrCost".
+;
+; That commit raised the cost of vector selects for non-throughput cost kinds,
+; causing SimplifyCFG to stop speculatively executing blocks on AMDGPU. Without
+; the speculation, the branch+PHI pattern creates additional register pressure
+; at the merge point, leading to SGPR spills.
+
+; ASM: .sgpr_spill_count: 8
+; ASM-THRESHOLD4: .sgpr_spill_count: 0
+
+define amdgpu_kernel void @reduced_fmha_kernel(i32 %arg, i32 %arg1, i32 %arg2, i1 %arg3, i32 %arg4, i1 %arg5, i1 %arg6, i1 %arg7, i1 %arg8, i32 %arg9, i1 %arg10, <4 x float> %arg11, i32 %arg12, i32 %arg13, i1 %arg14, i1 %arg15, i1 %arg16, i1 %arg17, i32 %arg18, i1 %arg19, i1 %arg20, i32 %arg21, i1 %arg22, i32 %arg23, i1 %arg24, i1 %arg25, i1 %arg26, i32 %arg27, i32 %arg28, i1 %arg29, i32 %arg30, i1 %arg31, i32 %arg32, i1 %arg33, i32 %arg34, <4 x float> %arg35, <4 x float> %arg36, <4 x float> %arg37) {
+; IR-LABEL: define amdgpu_kernel void @reduced_fmha_kernel(
+; IR-SAME: i32 [[ARG:%.*]], i32 [[ARG1:%.*]], i32 [[ARG2:%.*]], i1 [[ARG3:%.*]], i32 [[ARG4:%.*]], i1 [[ARG5:%.*]], i1 [[ARG6:%.*]], i1 [[ARG7:%.*]], i1 [[ARG8:%.*]], i32 [[ARG9:%.*]], i1 [[ARG10:%.*]], <4 x float> [[ARG11:%.*]], i32 [[ARG12:%.*]], i32 [[ARG13:%.*]], i1 [[ARG14:%.*]], i1 [[ARG15:%.*]], i1 [[ARG16:%.*]], i1 [[ARG17:%.*]], i32 [[ARG18:%.*]], i1 [[ARG19:%.*]], i1 [[ARG20:%.*]], i32 [[ARG21:%.*]], i1 [[ARG22:%.*]], i32 [[ARG23:%.*]], i1 [[ARG24:%.*]], i1 [[ARG25:%.*]], i1 [[ARG26:%.*]], i32 [[ARG27:%.*]], i32 [[ARG28:%.*]], i1 [[ARG29:%.*]], i32 [[ARG30:%.*]], i1 [[ARG31:%.*]], i32 [[ARG32:%.*]], i1 [[ARG33:%.*]], i32 [[ARG34:%.*]], <4 x float> [[ARG35:%.*]], <4 x float> [[ARG36:%.*]], <4 x float> [[ARG37:%.*]]) #[[ATTR0:[0-9]+]] {
+; IR-NEXT: [[BB:.*]]:
+; IR-NEXT: br label %[[BB38:.*]]
+; IR: [[BB38]]:
+; IR-NEXT: [[PHI:%.*]] = phi float [ 0.000000e+00, %[[BB]] ], [ [[FMUL102:%.*]], %[[BB86:.*]] ]
+; IR-NEXT: [[PHI39:%.*]] = phi <4 x float> [ zeroinitializer, %[[BB]] ], [ [[CALL105:%.*]], %[[BB86]] ]
+; IR-NEXT: br i1 [[ARG3]], label %[[BB40:.*]], label %[[BB86]]
+; IR: [[BB40]]:
+; IR-NEXT: [[CALL:%.*]] = tail call i32 @llvm.amdgcn.mbcnt.hi(i32 0, i32 0)
+; IR-NEXT: [[ICMP:%.*]] = icmp sge i32 [[ARG9]], [[CALL]]
+; IR-NEXT: [[OR:%.*]] = or i1 [[ARG10]], [[ICMP]]
+; IR-NEXT: [[CALL41:%.*]] = tail call i32 @llvm.amdgcn.readfirstlane.i32(i32 0)
+; IR-NEXT: [[ICMP42:%.*]] = icmp sge i32 [[ARG18]], [[CALL41]]
+; IR-NEXT: [[OR43:%.*]] = or i1 [[ARG19]], [[ICMP42]]
+; IR-NEXT: [[SELECT:%.*]] = select i1 [[OR43]], i1 false, i1 [[ARG3]]
+; IR-NEXT: [[SELECT44:%.*]] = select i1 [[OR]], <4 x float> zeroinitializer, <4 x float> [[ARG11]]
+; IR-NEXT: [[SELECT45:%.*]] = select i1 [[ARG14]], <4 x float> zeroinitializer, <4 x float> [[SELECT44]]
+; IR-NEXT: [[SELECT46:%.*]] = select i1 [[SELECT]], <4 x float> zeroinitializer, <4 x float> [[SELECT45]]
+; IR-NEXT: [[ICMP47:%.*]] = icmp sle i32 [[ARG21]], [[CALL]]
+; IR-NEXT: [[OR48:%.*]] = or i1 [[ICMP47]], [[ARG22]]
+; IR-NEXT: [[SELECT49:%.*]] = select i1 [[OR48]], <4 x float> splat (float 1.000000e+00), <4 x float> zeroinitializer
+; IR-NEXT: [[ICMP50:%.*]] = icmp sge i32 [[ARG23]], [[CALL41]]
+; IR-NEXT: [[ICMP51:%.*]] = icmp sle i32 [[ARG13]], [[CALL41]]
+; IR-NEXT: [[OR52:%.*]] = or i1 [[ICMP51]], [[ARG29]]
+; IR-NEXT: [[OR53:%.*]] = or i1 [[ARG8]], [[ICMP50]]
+; IR-NEXT: [[SELECT54:%.*]] = select i1 [[ARG25]], <4 x float> zeroinitializer, <4 x float> [[SELECT49]]
+; IR-NEXT: [[SELECT55:%.*]] = select i1 [[OR53]], <4 x float> zeroinitializer, <4 x float> [[SELECT54]]
+; IR-NEXT: [[SELECT56:%.*]] = select i1 [[OR52]], <4 x float> zeroinitializer, <4 x float> [[SELECT55]]
+; IR-NEXT: [[ICMP57:%.*]] = icmp sge i32 [[ARG30]], [[CALL]]
+; IR-NEXT: [[OR58:%.*]] = or i1 [[ARG6]], [[ICMP57]]
+; IR-NEXT: [[ICMP59:%.*]] = icmp sle i32 [[ARG34]], [[CALL41]]
+; IR-NEXT: [[OR60:%.*]] = or i1 [[ICMP59]], [[ARG15]]
+; IR-NEXT: [[SELECT61:%.*]] = select i1 [[OR58]], <4 x float> splat (float 1.000000e+00), <4 x float> zeroinitializer
+; IR-NEXT: [[SELECT62:%.*]] = select i1 [[ARG5]], <4 x float> zeroinitializer, <4 x float> [[SELECT61]]
+; IR-NEXT: [[SELECT63:%.*]] = select i1 [[OR60]], <4 x float> zeroinitializer, <4 x float> [[SELECT62]]
+; IR-NEXT: [[ICMP64:%.*]] = icmp sle i32 [[ARG]], [[CALL]]
+; IR-NEXT: [[ICMP65:%.*]] = icmp sge i32 [[ARG28]], [[CALL41]]
+; IR-NEXT: [[OR66:%.*]] = or i1 [[ARG33]], [[ICMP65]]
+; IR-NEXT: [[OR67:%.*]] = or i1 [[ICMP64]], [[ARG17]]
+; IR-NEXT: [[SELECT68:%.*]] = select i1 [[OR67]], <4 x float> zeroinitializer, <4 x float> [[ARG35]]
+; IR-NEXT: [[SELECT69:%.*]] = select i1 [[OR66]], <4 x float> zeroinitializer, <4 x float> [[SELECT68]]
+; IR-NEXT: [[SELECT70:%.*]] = select i1 [[ARG7]], <4 x float> zeroinitializer, <4 x float> [[SELECT69]]
+; IR-NEXT: [[ICMP71:%.*]] = icmp sle i32 [[ARG32]], [[CALL]]
+; IR-NEXT: [[ICMP72:%.*]] = icmp sge i32 [[ARG2]], [[CALL41]]
+; IR-NEXT: [[OR73:%.*]] = or i1 [[ARG20]], [[ICMP72]]
+; IR-NEXT: [[OR74:%.*]] = or i1 [[ICMP71]], [[ARG26]]
+; IR-NEXT: [[SELECT75:%.*]] = select i1 [[OR74]], <4 x float> splat (float 1.000000e+00), <4 x float> zeroinitializer
+; IR-NEXT: [[SELECT76:%.*]] = select i1 [[OR73]], <4 x float> zeroinitializer, <4 x float> [[SELECT75]]
+; IR-NEXT: [[SELECT77:%.*]] = select i1 [[ARG16]], <4 x float> zeroinitializer, <4 x float> [[SELECT76]]
+; IR-NEXT: [[ICMP78:%.*]] = icmp sge i32 [[ARG4]], [[CALL41]]
+; IR-NEXT: [[OR79:%.*]] = or i1 [[ARG24]], [[ICMP78]]
+; IR-NEXT: [[SELECT80:%.*]] = select i1 [[OR79]], <4 x float> zeroinitializer, <4 x float> [[ARG36]]
+; IR-NEXT: [[SELECT81:%.*]] = select i1 [[ARG5]], <4 x float> zeroinitializer, <4 x float> [[SELECT80]]
+; IR-NEXT: [[SELECT82:%.*]] = select i1 [[ARG6]], <4 x float> zeroinitializer, <4 x float> [[SELECT81]]
+; IR-NEXT: [[OR83:%.*]] = or i32 [[ARG1]], [[CALL]]
+; IR-NEXT: [[ICMP84:%.*]] = icmp sge i32 [[OR83]], 0
+; IR-NEXT: br i1 [[ICMP84]], label %[[BB85:.*]], label %[[BB86]]
+; IR: [[BB85]]:
+; IR-NEXT: br label %[[BB86]]
+; IR: [[BB86]]:
+; IR-NEXT: [[PHI87:%.*]] = phi <4 x float> [ [[SELECT46]], %[[BB85]] ], [ [[SELECT46]], %[[BB40]] ], [ zeroinitializer, %[[BB38]] ]
+; IR-NEXT: [[PHI88:%.*]] = phi <4 x float> [ [[SELECT56]], %[[BB85]] ], [ [[SELECT56]], %[[BB40]] ], [ zeroinitializer, %[[BB38]] ]
+; IR-NEXT: [[PHI89:%.*]] = phi <4 x float> [ [[SELECT63]], %[[BB85]] ], [ [[SELECT63]], %[[BB40]] ], [ zeroinitializer, %[[BB38]] ]
+; IR-NEXT: [[PHI90:%.*]] = phi <4 x float> [ [[SELECT70]], %[[BB85]] ], [ [[SELECT70]], %[[BB40]] ], [ zeroinitializer, %[[BB38]] ]
+; IR-NEXT: [[PHI91:%.*]] = phi <4 x float> [ [[SELECT77]], %[[BB85]] ], [ [[SELECT77]], %[[BB40]] ], [ zeroinitializer, %[[BB38]] ]
+; IR-NEXT: [[PHI92:%.*]] = phi <4 x float> [ [[SELECT82]], %[[BB85]] ], [ [[ARG37]], %[[BB40]] ], [ zeroinitializer, %[[BB38]] ]
+; IR-NEXT: [[EXTRACTELEMENT:%.*]] = extractelement <4 x float> [[PHI87]], i64 0
+; IR-NEXT: [[EXTRACTELEMENT93:%.*]] = extractelement <4 x float> [[PHI88]], i64 0
+; IR-NEXT: [[FMUL:%.*]] = fmul float [[EXTRACTELEMENT]], [[EXTRACTELEMENT93]]
+; IR-NEXT: [[EXTRACTELEMENT94:%.*]] = extractelement <4 x float> [[PHI89]], i64 0
+; IR-NEXT: [[FMUL95:%.*]] = fmul float [[FMUL]], [[EXTRACTELEMENT94]]
+; IR-NEXT: [[EXTRACTELEMENT96:%.*]] = extractelement <4 x float> [[PHI90]], i64 0
+; IR-NEXT: [[FMUL97:%.*]] = fmul float [[FMUL95]], [[EXTRACTELEMENT96]]
+; IR-NEXT: [[EXTRACTELEMENT98:%.*]] = extractelement <4 x float> [[PHI91]], i64 0
+; IR-NEXT: [[FMUL99:%.*]] = fmul float [[FMUL97]], [[EXTRACTELEMENT98]]
+; IR-NEXT: [[EXTRACTELEMENT100:%.*]] = extractelement <4 x float> [[PHI92]], i64 0
+; IR-NEXT: [[FMUL101:%.*]] = fmul float [[FMUL99]], [[EXTRACTELEMENT100]]
+; IR-NEXT: [[FMUL102]] = fmul float [[PHI]], [[FMUL101]]
+; IR-NEXT: [[EXTRACTELEMENT103:%.*]] = extractelement <4 x float> [[PHI39]], i64 0
+; IR-NEXT: [[FMUL104:%.*]] = fmul float [[EXTRACTELEMENT103]], 0.000000e+00
+; IR-NEXT: [[FPTRUNC:%.*]] = fptrunc float [[PHI]] to half
+; IR-NEXT: [[INSERTELEMENT:%.*]] = insertelement <4 x half> zeroinitializer, half [[FPTRUNC]], i64 0
+; IR-NEXT: [[CALL105]] = tail call <4 x float> @llvm.amdgcn.mfma.f32.16x16x16f16(<4 x half> zeroinitializer, <4 x half> [[INSERTELEMENT]], <4 x float> zeroinitializer, i32 0, i32 0, i32 0)
+; IR-NEXT: br label %[[BB38]]
+;
+bb:
+ br label %bb38
+
+bb38:
+ %phi = phi float [ 0.000000e+00, %bb ], [ %fmul102, %bb86 ]
+ %phi39 = phi <4 x float> [ zeroinitializer, %bb ], [ %call105, %bb86 ]
+ br i1 %arg3, label %bb40, label %bb86
+
+bb40:
+ %call = tail call i32 @llvm.amdgcn.mbcnt.hi(i32 0, i32 0)
+ %icmp = icmp sge i32 %arg9, %call
+ %or = or i1 %arg10, %icmp
+ %call41 = tail call i32 @llvm.amdgcn.readfirstlane.i32(i32 0)
+ %icmp42 = icmp sge i32 %arg18, %call41
+ %or43 = or i1 %arg19, %icmp42
+ %select = select i1 %or43, i1 false, i1 %arg3
+ %select44 = select i1 %or, <4 x float> zeroinitializer, <4 x float> %arg11
+ %select45 = select i1 %arg14, <4 x float> zeroinitializer, <4 x float> %select44
+ %select46 = select i1 %select, <4 x float> zeroinitializer, <4 x float> %select45
+ %icmp47 = icmp sle i32 %arg21, %call
+ %or48 = or i1 %icmp47, %arg22
+ %select49 = select i1 %or48, <4 x float> splat (float 1.000000e+00), <4 x float> zeroinitializer
+ %icmp50 = icmp sge i32 %arg23, %call41
+ %icmp51 = icmp sle i32 %arg13, %call41
+ %or52 = or i1 %icmp51, %arg29
+ %or53 = or i1 %arg8, %icmp50
+ %select54 = select i1 %arg25, <4 x float> zeroinitializer, <4 x float> %select49
+ %select55 = select i1 %or53, <4 x float> zeroinitializer, <4 x float> %select54
+ %select56 = select i1 %or52, <4 x float> zeroinitializer, <4 x float> %select55
+ %icmp57 = icmp sge i32 %arg30, %call
+ %or58 = or i1 %arg6, %icmp57
+ %icmp59 = icmp sle i32 %arg34, %call41
+ %or60 = or i1 %icmp59, %arg15
+ %select61 = select i1 %or58, <4 x float> splat (float 1.000000e+00), <4 x float> zeroinitializer
+ %select62 = select i1 %arg5, <4 x float> zeroinitializer, <4 x float> %select61
+ %select63 = select i1 %or60, <4 x float> zeroinitializer, <4 x float> %select62
+ %icmp64 = icmp sle i32 %arg, %call
+ %icmp65 = icmp sge i32 %arg28, %call41
+ %or66 = or i1 %arg33, %icmp65
+ %or67 = or i1 %icmp64, %arg17
+ %select68 = select i1 %or67, <4 x float> zeroinitializer, <4 x float> %arg35
+ %select69 = select i1 %or66, <4 x float> zeroinitializer, <4 x float> %select68
+ %select70 = select i1 %arg7, <4 x float> zeroinitializer, <4 x float> %select69
+ %icmp71 = icmp sle i32 %arg32, %call
+ %icmp72 = icmp sge i32 %arg2, %call41
+ %or73 = or i1 %arg20, %icmp72
+ %or74 = or i1 %icmp71, %arg26
+ %select75 = select i1 %or74, <4 x float> splat (float 1.000000e+00), <4 x float> zeroinitializer
+ %select76 = select i1 %or73, <4 x float> zeroinitializer, <4 x float> %select75
+ %select77 = select i1 %arg16, <4 x float> zeroinitializer, <4 x float> %select76
+ %icmp78 = icmp sge i32 %arg4, %call41
+ %or79 = or i1 %arg24, %icmp78
+ %select80 = select i1 %or79, <4 x float> zeroinitializer, <4 x float> %arg36
+ %select81 = select i1 %arg5, <4 x float> zeroinitializer, <4 x float> %select80
+ %select82 = select i1 %arg6, <4 x float> zeroinitializer, <4 x float> %select81
+ %or83 = or i32 %arg1, %call
+ %icmp84 = icmp sge i32 %or83, 0
+ br i1 %icmp84, label %bb85, label %bb86
+
+bb85:
+ br label %bb86
+
+bb86:
+ %phi87 = phi <4 x float> [ %select46, %bb85 ], [ %select46, %bb40 ], [ zeroinitializer, %bb38 ]
+ %phi88 = phi <4 x float> [ %select56, %bb85 ], [ %select56, %bb40 ], [ zeroinitializer, %bb38 ]
+ %phi89 = phi <4 x float> [ %select63, %bb85 ], [ %select63, %bb40 ], [ zeroinitializer, %bb38 ]
+ %phi90 = phi <4 x float> [ %select70, %bb85 ], [ %select70, %bb40 ], [ zeroinitializer, %bb38 ]
+ %phi91 = phi <4 x float> [ %select77, %bb85 ], [ %select77, %bb40 ], [ zeroinitializer, %bb38 ]
+ %phi92 = phi <4 x float> [ %select82, %bb85 ], [ %arg37, %bb40 ], [ zeroinitializer, %bb38 ]
+ %extractelement = extractelement <4 x float> %phi87, i64 0
+ %extractelement93 = extractelement <4 x float> %phi88, i64 0
+ %fmul = fmul float %extractelement, %extractelement93
+ %extractelement94 = extractelement <4 x float> %phi89, i64 0
+ %fmul95 = fmul float %fmul, %extractelement94
+ %extractelement96 = extractelement <4 x float> %phi90, i64 0
+ %fmul97 = fmul float %fmul95, %extractelement96
+ %extractelement98 = extractelement <4 x float> %phi91, i64 0
+ %fmul99 = fmul float %fmul97, %extractelement98
+ %extractelement100 = extractelement <4 x float> %phi92, i64 0
+ %fmul101 = fmul float %fmul99, %extractelement100
+ %fmul102 = fmul float %phi, %fmul101
+ %extractelement103 = extractelement <4 x float> %phi39, i64 0
+ %fmul104 = fmul float %extractelement103, 0.000000e+00
+ %fptrunc = fptrunc float %phi to half
+ %insertelement = insertelement <4 x half> zeroinitializer, half %fptrunc, i64 0
+ %call105 = tail call <4 x float> @llvm.amdgcn.mfma.f32.16x16x16f16(<4 x half> zeroinitializer, <4 x half> %insertelement, <4 x float> zeroinitializer, i32 0, i32 0, i32 0)
+ br label %bb38
+}
+
+declare i32 @llvm.amdgcn.readfirstlane.i32(i32)
+declare i32 @llvm.amdgcn.mbcnt.hi(i32, i32)
+declare <4 x float> @llvm.amdgcn.mfma.f32.16x16x16f16(<4 x half>, <4 x half>, <4 x float>, i32, i32, i32)
diff --git a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll
new file mode 100644
index 0000000000000..4d889569f64ec
--- /dev/null
+++ b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll
@@ -0,0 +1,44 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa < %s | FileCheck %s --check-prefix=DEFAULT
+; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa -phi-node-folding-threshold=4 < %s | FileCheck %s --check-prefix=THRESHOLD4
+
+; Test that SimplifyCFG speculatively executes blocks on AMDGPU even for vector
+; types. The base cost model returns cost 4 for a <4 x float> select, so the
+; block is not speculated with the default threshold. Raising the threshold
+; to 4 allows the speculation.
+
+define amdgpu_kernel void @ham(i1 %arg3, i1 %icmp84) {
+; DEFAULT-LABEL: define amdgpu_kernel void @ham(
+; DEFAULT-SAME: i1 [[ARG3:%.*]], i1 [[ICMP84:%.*]]) {
+; DEFAULT-NEXT: [[BB:.*]]:
+; DEFAULT-NEXT: br i1 [[ARG3]], label %[[BB40:.*]], label %[[BB86:.*]]
+; DEFAULT: [[BB40]]:
+; DEFAULT-NEXT: [[ICMP841:%.*]] = icmp sge i32 0, 0
+; DEFAULT-NEXT: br i1 [[ICMP84]], label %[[BB85:.*]], label %[[BB86]]
+; DEFAULT: [[BB85]]:
+; DEFAULT-NEXT: br label %[[BB86]]
+; DEFAULT: [[BB86]]:
+; DEFAULT-NEXT: [[PHI92:%.*]] = phi <4 x float> [ zeroinitializer, %[[BB85]] ], [ splat (float 1.000000e+00), %[[BB40]] ], [ zeroinitializer, %[[BB]] ]
+; DEFAULT-NEXT: ret void
+;
+; THRESHOLD4-LABEL: define amdgpu_kernel void @ham(
+; THRESHOLD4-SAME: i1 [[ARG3:%.*]], i1 [[ICMP84:%.*]]) {
+; THRESHOLD4-NEXT: [[BB:.*:]]
+; THRESHOLD4-NEXT: [[SPEC_SELECT:%.*]] = select i1 [[ICMP84]], <4 x float> zeroinitializer, <4 x float> splat (float 1.000000e+00)
+; THRESHOLD4-NEXT: [[SPEC_SELECT1:%.*]] = select i1 [[ARG3]], <4 x float> [[SPEC_SELECT]], <4 x float> zeroinitializer
+; THRESHOLD4-NEXT: ret void
+;
+entry:
+ br i1 %arg3, label %then, label %merge
+
+then: ; preds = %entry
+ %icmp841 = icmp sge i32 0, 0
+ br i1 %icmp84, label %then.if, label %merge
+
+then.if: ; preds = %then
+ br label %merge
+
+merge: ; preds = %then.if, %then, %entry
+ %phi92 = phi <4 x float> [ zeroinitializer, %then.if ], [ splat (float 1.000000e+00), %then ], [ zeroinitializer, %entry ]
+ ret void
+}
>From e559b4256e9e0b7825b0434bdb684650c516e8cf Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Fri, 3 Jul 2026 12:30:03 -0400
Subject: [PATCH 02/20] [AMDGPU] Return low select cost for non-throughput cost
kinds
Commit 0967957d7a94 ("[CostModel] Handle all cost kinds in
getCmpSelInstrCost") changed the base cost model to return type
legalization costs for non-throughput cost kinds. This caused
SimplifyCFG to stop folding branches to selects on AMDGPU, since
vector selects now report higher costs that exceed the folding
threshold.
Add an getCmpSelInstrCost override for AMDGPU to restore the old
behavior. This cost kind does just affect the SimplifyCFG
block speculation which is important and there seems to be
no good possibility to make a more precise decision at
this point.
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 17 ++++++++++
.../Target/AMDGPU/AMDGPUTargetTransformInfo.h | 7 ++++
.../AMDGPU/speculate-block-regpressure.ll | 17 +++++-----
.../SimplifyCFG/AMDGPU/speculate-block.ll | 33 +++++--------------
4 files changed, 41 insertions(+), 33 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 7631bb2dc6828..9925df32e1892 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -979,6 +979,23 @@ InstructionCost GCNTTIImpl::getCFInstrCost(unsigned Opcode,
return BaseT::getCFInstrCost(Opcode, CostKind, I);
}
+InstructionCost GCNTTIImpl::getCmpSelInstrCost(
+ unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
+ TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info,
+ TTI::OperandValueInfo Op2Info, const Instruction *I) const {
+ // FIXME: Commit 0967957d7a94 changed the base implementation to return a
+ // cost that scales with vector width for non-throughput cost kinds. This
+ // causes SimplifyCFG's speculativelyExecuteBB (which uses TCK_SizeAndLatency)
+ // to stop folding branches to selects. We want the fold for AMDGPU, and
+ // reverting to the pre-0967957d7a94 behavior seems appropriate since
+ // SimplifyCFG lacks the context for a thorough profitability analysis.
+ if (CostKind != TTI::TCK_RecipThroughput)
+ return 1;
+
+ return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
+ Op1Info, Op2Info, I);
+}
+
InstructionCost
GCNTTIImpl::getArithmeticReductionCost(unsigned Opcode, VectorType *Ty,
std::optional<FastMathFlags> FMF,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
index 4a239f4d6983d..647b11e2098bb 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
@@ -175,6 +175,13 @@ class GCNTTIImpl final : public BasicTTIImplBase<GCNTTIImpl> {
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind,
const Instruction *I = nullptr) const override;
+ InstructionCost getCmpSelInstrCost(
+ unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
+ TTI::TargetCostKind CostKind,
+ TTI::OperandValueInfo Op1Info = {TTI::OK_AnyValue, TTI::OP_None},
+ TTI::OperandValueInfo Op2Info = {TTI::OK_AnyValue, TTI::OP_None},
+ const Instruction *I = nullptr) const override;
+
bool isInlineAsmSourceOfDivergence(const CallInst *CI,
ArrayRef<unsigned> Indices = {}) const;
diff --git a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
index 91914bec8cdb0..8e83438b154ee 100644
--- a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
+++ b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
@@ -13,7 +13,7 @@
; the speculation, the branch+PHI pattern creates additional register pressure
; at the merge point, leading to SGPR spills.
-; ASM: .sgpr_spill_count: 8
+; ASM: .sgpr_spill_count: 0
; ASM-THRESHOLD4: .sgpr_spill_count: 0
define amdgpu_kernel void @reduced_fmha_kernel(i32 %arg, i32 %arg1, i32 %arg2, i1 %arg3, i32 %arg4, i1 %arg5, i1 %arg6, i1 %arg7, i1 %arg8, i32 %arg9, i1 %arg10, <4 x float> %arg11, i32 %arg12, i32 %arg13, i1 %arg14, i1 %arg15, i1 %arg16, i1 %arg17, i32 %arg18, i1 %arg19, i1 %arg20, i32 %arg21, i1 %arg22, i32 %arg23, i1 %arg24, i1 %arg25, i1 %arg26, i32 %arg27, i32 %arg28, i1 %arg29, i32 %arg30, i1 %arg31, i32 %arg32, i1 %arg33, i32 %arg34, <4 x float> %arg35, <4 x float> %arg36, <4 x float> %arg37) {
@@ -74,16 +74,15 @@ define amdgpu_kernel void @reduced_fmha_kernel(i32 %arg, i32 %arg1, i32 %arg2, i
; IR-NEXT: [[SELECT82:%.*]] = select i1 [[ARG6]], <4 x float> zeroinitializer, <4 x float> [[SELECT81]]
; IR-NEXT: [[OR83:%.*]] = or i32 [[ARG1]], [[CALL]]
; IR-NEXT: [[ICMP84:%.*]] = icmp sge i32 [[OR83]], 0
-; IR-NEXT: br i1 [[ICMP84]], label %[[BB85:.*]], label %[[BB86]]
-; IR: [[BB85]]:
+; IR-NEXT: [[SPEC_SELECT:%.*]] = select i1 [[ICMP84]], <4 x float> [[SELECT82]], <4 x float> [[ARG37]]
; IR-NEXT: br label %[[BB86]]
; IR: [[BB86]]:
-; IR-NEXT: [[PHI87:%.*]] = phi <4 x float> [ [[SELECT46]], %[[BB85]] ], [ [[SELECT46]], %[[BB40]] ], [ zeroinitializer, %[[BB38]] ]
-; IR-NEXT: [[PHI88:%.*]] = phi <4 x float> [ [[SELECT56]], %[[BB85]] ], [ [[SELECT56]], %[[BB40]] ], [ zeroinitializer, %[[BB38]] ]
-; IR-NEXT: [[PHI89:%.*]] = phi <4 x float> [ [[SELECT63]], %[[BB85]] ], [ [[SELECT63]], %[[BB40]] ], [ zeroinitializer, %[[BB38]] ]
-; IR-NEXT: [[PHI90:%.*]] = phi <4 x float> [ [[SELECT70]], %[[BB85]] ], [ [[SELECT70]], %[[BB40]] ], [ zeroinitializer, %[[BB38]] ]
-; IR-NEXT: [[PHI91:%.*]] = phi <4 x float> [ [[SELECT77]], %[[BB85]] ], [ [[SELECT77]], %[[BB40]] ], [ zeroinitializer, %[[BB38]] ]
-; IR-NEXT: [[PHI92:%.*]] = phi <4 x float> [ [[SELECT82]], %[[BB85]] ], [ [[ARG37]], %[[BB40]] ], [ zeroinitializer, %[[BB38]] ]
+; IR-NEXT: [[PHI87:%.*]] = phi <4 x float> [ zeroinitializer, %[[BB38]] ], [ [[SELECT46]], %[[BB40]] ]
+; IR-NEXT: [[PHI88:%.*]] = phi <4 x float> [ zeroinitializer, %[[BB38]] ], [ [[SELECT56]], %[[BB40]] ]
+; IR-NEXT: [[PHI89:%.*]] = phi <4 x float> [ zeroinitializer, %[[BB38]] ], [ [[SELECT63]], %[[BB40]] ]
+; IR-NEXT: [[PHI90:%.*]] = phi <4 x float> [ zeroinitializer, %[[BB38]] ], [ [[SELECT70]], %[[BB40]] ]
+; IR-NEXT: [[PHI91:%.*]] = phi <4 x float> [ zeroinitializer, %[[BB38]] ], [ [[SELECT77]], %[[BB40]] ]
+; IR-NEXT: [[PHI92:%.*]] = phi <4 x float> [ zeroinitializer, %[[BB38]] ], [ [[SPEC_SELECT]], %[[BB40]] ]
; IR-NEXT: [[EXTRACTELEMENT:%.*]] = extractelement <4 x float> [[PHI87]], i64 0
; IR-NEXT: [[EXTRACTELEMENT93:%.*]] = extractelement <4 x float> [[PHI88]], i64 0
; IR-NEXT: [[FMUL:%.*]] = fmul float [[EXTRACTELEMENT]], [[EXTRACTELEMENT93]]
diff --git a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll
index 4d889569f64ec..37327b63c613a 100644
--- a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll
+++ b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll
@@ -1,32 +1,17 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
-; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa < %s | FileCheck %s --check-prefix=DEFAULT
-; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa -phi-node-folding-threshold=4 < %s | FileCheck %s --check-prefix=THRESHOLD4
+; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa < %s | FileCheck %s
; Test that SimplifyCFG speculatively executes blocks on AMDGPU even for vector
-; types. The base cost model returns cost 4 for a <4 x float> select, so the
-; block is not speculated with the default threshold. Raising the threshold
-; to 4 allows the speculation.
+; types. The AMDGPU getCmpSelInstrCost override returns a low cost for selects,
+; allowing the speculation regardless of vector width.
define amdgpu_kernel void @ham(i1 %arg3, i1 %icmp84) {
-; DEFAULT-LABEL: define amdgpu_kernel void @ham(
-; DEFAULT-SAME: i1 [[ARG3:%.*]], i1 [[ICMP84:%.*]]) {
-; DEFAULT-NEXT: [[BB:.*]]:
-; DEFAULT-NEXT: br i1 [[ARG3]], label %[[BB40:.*]], label %[[BB86:.*]]
-; DEFAULT: [[BB40]]:
-; DEFAULT-NEXT: [[ICMP841:%.*]] = icmp sge i32 0, 0
-; DEFAULT-NEXT: br i1 [[ICMP84]], label %[[BB85:.*]], label %[[BB86]]
-; DEFAULT: [[BB85]]:
-; DEFAULT-NEXT: br label %[[BB86]]
-; DEFAULT: [[BB86]]:
-; DEFAULT-NEXT: [[PHI92:%.*]] = phi <4 x float> [ zeroinitializer, %[[BB85]] ], [ splat (float 1.000000e+00), %[[BB40]] ], [ zeroinitializer, %[[BB]] ]
-; DEFAULT-NEXT: ret void
-;
-; THRESHOLD4-LABEL: define amdgpu_kernel void @ham(
-; THRESHOLD4-SAME: i1 [[ARG3:%.*]], i1 [[ICMP84:%.*]]) {
-; THRESHOLD4-NEXT: [[BB:.*:]]
-; THRESHOLD4-NEXT: [[SPEC_SELECT:%.*]] = select i1 [[ICMP84]], <4 x float> zeroinitializer, <4 x float> splat (float 1.000000e+00)
-; THRESHOLD4-NEXT: [[SPEC_SELECT1:%.*]] = select i1 [[ARG3]], <4 x float> [[SPEC_SELECT]], <4 x float> zeroinitializer
-; THRESHOLD4-NEXT: ret void
+; CHECK-LABEL: define amdgpu_kernel void @ham(
+; CHECK-SAME: i1 [[ARG3:%.*]], i1 [[ICMP84:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[SPEC_SELECT:%.*]] = select i1 [[ICMP84]], <4 x float> zeroinitializer, <4 x float> splat (float 1.000000e+00)
+; CHECK-NEXT: [[SPEC_SELECT1:%.*]] = select i1 [[ARG3]], <4 x float> [[SPEC_SELECT]], <4 x float> zeroinitializer
+; CHECK-NEXT: ret void
;
entry:
br i1 %arg3, label %then, label %merge
>From 3adec7f2c17ff52cf216f908c55f43d3c9cb05dc Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Wed, 8 Jul 2026 06:39:14 -0400
Subject: [PATCH 03/20] Update test checks
---
llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll | 4 ++--
llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll | 4 ++--
2 files changed, 4 insertions(+), 4 deletions(-)
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll b/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll
index 00ce9680cab32..781eba144205d 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll
@@ -24,8 +24,8 @@ define i32 @reduce_i1(i32 %arg) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V16 = call i1 @llvm.vector.reduce.and.v16i1(<16 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V32 = call i1 @llvm.vector.reduce.and.v32i1(<32 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V64 = call i1 @llvm.vector.reduce.and.v64i1(<64 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 130 for instruction: %V128 = call i1 @llvm.vector.reduce.and.v128i1(<128 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 260 for instruction: %V256 = call i1 @llvm.vector.reduce.and.v256i1(<256 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V128 = call i1 @llvm.vector.reduce.and.v128i1(<128 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 257 for instruction: %V256 = call i1 @llvm.vector.reduce.and.v256i1(<256 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
;
%V1 = call i1 @llvm.vector.reduce.and.v1i1(<1 x i1> undef)
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll b/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll
index eccaffc9d2a40..9e03620e2aa6f 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll
@@ -24,8 +24,8 @@ define i32 @reduce_i1(i32 %arg) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V16 = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V32 = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V64 = call i1 @llvm.vector.reduce.or.v64i1(<64 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 130 for instruction: %V128 = call i1 @llvm.vector.reduce.or.v128i1(<128 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 260 for instruction: %V256 = call i1 @llvm.vector.reduce.or.v256i1(<256 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V128 = call i1 @llvm.vector.reduce.or.v128i1(<128 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 257 for instruction: %V256 = call i1 @llvm.vector.reduce.or.v256i1(<256 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
;
%V1 = call i1 @llvm.vector.reduce.or.v1i1(<1 x i1> undef)
>From 860aa479872b8e14d315c22737f88a294b07ca59 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Wed, 8 Jul 2026 09:20:04 -0400
Subject: [PATCH 04/20] Remove additional test run
---
.../AMDGPU/speculate-block-regpressure.ll | 14 +++++---------
1 file changed, 5 insertions(+), 9 deletions(-)
diff --git a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
index 8e83438b154ee..818fbc7928188 100644
--- a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
+++ b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
@@ -2,19 +2,15 @@
; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a < %s | FileCheck %s --check-prefix=IR
; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a < %s \
; RUN: | llc -O3 -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a | FileCheck %s --check-prefix=ASM
-; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -phi-node-folding-threshold=4 < %s \
-; RUN: | llc -O3 -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a | FileCheck %s --check-prefix=ASM-THRESHOLD4
; Regression test for SGPR spilling caused by commit 0967957d7a94
-; "[CostModel] Handle all cost kinds in getCmpSelInstrCost".
-;
-; That commit raised the cost of vector selects for non-throughput cost kinds,
-; causing SimplifyCFG to stop speculatively executing blocks on AMDGPU. Without
-; the speculation, the branch+PHI pattern creates additional register pressure
-; at the merge point, leading to SGPR spills.
+; "[CostModel] Handle all cost kinds in getCmpSelInstrCost". That
+; commit raised the cost of vector selects for non-throughput cost
+; kinds, causing SimplifyCFG to stop speculatively executing blocks on
+; AMDGPU which increases register pressure at the merge point,
+; leading to SGPR spills.
; ASM: .sgpr_spill_count: 0
-; ASM-THRESHOLD4: .sgpr_spill_count: 0
define amdgpu_kernel void @reduced_fmha_kernel(i32 %arg, i32 %arg1, i32 %arg2, i1 %arg3, i32 %arg4, i1 %arg5, i1 %arg6, i1 %arg7, i1 %arg8, i32 %arg9, i1 %arg10, <4 x float> %arg11, i32 %arg12, i32 %arg13, i1 %arg14, i1 %arg15, i1 %arg16, i1 %arg17, i32 %arg18, i1 %arg19, i1 %arg20, i32 %arg21, i1 %arg22, i32 %arg23, i1 %arg24, i1 %arg25, i1 %arg26, i32 %arg27, i32 %arg28, i1 %arg29, i32 %arg30, i1 %arg31, i32 %arg32, i1 %arg33, i32 %arg34, <4 x float> %arg35, <4 x float> %arg36, <4 x float> %arg37) {
; IR-LABEL: define amdgpu_kernel void @reduced_fmha_kernel(
>From f0dfeab2714773123a5c7066f37ad306d688aeb7 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Thu, 9 Jul 2026 04:39:03 -0400
Subject: [PATCH 05/20] fixup! [AMDGPU] Return low select cost for
non-throughput cost kinds
---
llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp | 8 ++------
1 file changed, 2 insertions(+), 6 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 9925df32e1892..c6e6deb84d42c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -983,12 +983,8 @@ InstructionCost GCNTTIImpl::getCmpSelInstrCost(
unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info,
TTI::OperandValueInfo Op2Info, const Instruction *I) const {
- // FIXME: Commit 0967957d7a94 changed the base implementation to return a
- // cost that scales with vector width for non-throughput cost kinds. This
- // causes SimplifyCFG's speculativelyExecuteBB (which uses TCK_SizeAndLatency)
- // to stop folding branches to selects. We want the fold for AMDGPU, and
- // reverting to the pre-0967957d7a94 behavior seems appropriate since
- // SimplifyCFG lacks the context for a thorough profitability analysis.
+ // For size and latency cost kinds, return a low cost independent of vector
+ // width to enable SimplifyCFG's speculativelyExecuteBB optimization.
if (CostKind != TTI::TCK_RecipThroughput)
return 1;
>From ace3547ffc8005e11fb4281ee6ca108fbd721d50 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Thu, 9 Jul 2026 06:54:04 -0400
Subject: [PATCH 06/20] Stop calling base implementation from
getCmpSelInstrCost
This essentially duplicates the base implementation with some
simplifications (avoid recursive call for scalars, inline call to
TargetTransformInfoImpl::getCmpSelInstrCost).
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 40 ++++++++++++++++++-
1 file changed, 38 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index c6e6deb84d42c..6e4f30524b27f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -29,6 +29,7 @@
#include "llvm/IR/IntrinsicsAMDGPU.h"
#include "llvm/IR/PatternMatch.h"
#include "llvm/Support/KnownBits.h"
+#include <llvm/CodeGen/ISDOpcodes.h>
#include <optional>
using namespace llvm;
@@ -988,8 +989,43 @@ InstructionCost GCNTTIImpl::getCmpSelInstrCost(
if (CostKind != TTI::TCK_RecipThroughput)
return 1;
- return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
- Op1Info, Op2Info, I);
+ // Compute cost based on type legalization.
+ const TargetLoweringBase *TLI = getTLI();
+ if (TLI->getValueType(DL, ValTy, true) == MVT::Other)
+ return 1;
+
+ const int ISD = TLI->InstructionOpcodeToISD(Opcode);
+ assert(ISD && "Invalid opcode");
+ auto getScalarCost = [TLI, ISD](std::pair<InstructionCost, MVT> ScalarLT) {
+ return TLI->isOperationExpand(ISD, ScalarLT.second) ? 1 : ScalarLT.first;
+ };
+
+ const std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
+ if (!ValTy->isVectorTy())
+ return getScalarCost(LT);
+
+ auto *ValVTy = dyn_cast_or_null<FixedVectorType>(ValTy);
+ if (!ValVTy)
+ return InstructionCost::getInvalid();
+
+ // Return legalization cost for legal vector types.
+ // TODO Handle splitting as suggested but not implemented in the base
+ // implementation?
+ if (LT.second.isVector()) {
+ const int VecISD =
+ ISD == ISD::SELECT && CondTy->isVectorTy() ? ISD::VSELECT : ISD;
+ if (!TLI->isOperationExpand(VecISD, LT.second))
+ return LT.first;
+ }
+
+ // Otherwise, assume vector needs to be scalarized.
+ InstructionCost ScalarCost =
+ getScalarCost(getTypeLegalizationCost(ValVTy->getScalarType()));
+
+ // Return scalar cost plus cost of inserting all values.
+ return getScalarizationOverhead(ValVTy, /*Insert*/ true,
+ /*Extract*/ false, CostKind) +
+ ValVTy->getNumElements() * ScalarCost;
}
InstructionCost
>From 1bd3690441eb611541f12fd7d7620ac3a896e4b2 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Thu, 30 Jul 2026 10:05:41 -0400
Subject: [PATCH 07/20] Set 'Extract' to true in scalarization overhead
computation
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 47 +++++++++----------
.../SLPVectorizer/AMDGPU/min_max.ll | 13 +++--
2 files changed, 31 insertions(+), 29 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 6e4f30524b27f..00a38b3f143ea 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -24,10 +24,12 @@
#include "llvm/Analysis/LoopInfo.h"
#include "llvm/Analysis/ValueTracking.h"
#include "llvm/CodeGen/Analysis.h"
+#include "llvm/IR/DerivedTypes.h"
#include "llvm/IR/Function.h"
#include "llvm/IR/IRBuilder.h"
#include "llvm/IR/IntrinsicsAMDGPU.h"
#include "llvm/IR/PatternMatch.h"
+#include "llvm/Support/Casting.h"
#include "llvm/Support/KnownBits.h"
#include <llvm/CodeGen/ISDOpcodes.h>
#include <optional>
@@ -984,6 +986,9 @@ InstructionCost GCNTTIImpl::getCmpSelInstrCost(
unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info,
TTI::OperandValueInfo Op2Info, const Instruction *I) const {
+ if (isa<ScalableVectorType>(ValTy))
+ return InstructionCost::getInvalid();
+
// For size and latency cost kinds, return a low cost independent of vector
// width to enable SimplifyCFG's speculativelyExecuteBB optimization.
if (CostKind != TTI::TCK_RecipThroughput)
@@ -996,36 +1001,30 @@ InstructionCost GCNTTIImpl::getCmpSelInstrCost(
const int ISD = TLI->InstructionOpcodeToISD(Opcode);
assert(ISD && "Invalid opcode");
- auto getScalarCost = [TLI, ISD](std::pair<InstructionCost, MVT> ScalarLT) {
- return TLI->isOperationExpand(ISD, ScalarLT.second) ? 1 : ScalarLT.first;
- };
+ std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
- const std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
if (!ValTy->isVectorTy())
- return getScalarCost(LT);
+ return TLI->isOperationExpand(ISD, LT.second) ? 1 : LT.first;
- auto *ValVTy = dyn_cast_or_null<FixedVectorType>(ValTy);
- if (!ValVTy)
- return InstructionCost::getInvalid();
-
- // Return legalization cost for legal vector types.
- // TODO Handle splitting as suggested but not implemented in the base
- // implementation?
- if (LT.second.isVector()) {
- const int VecISD =
- ISD == ISD::SELECT && CondTy->isVectorTy() ? ISD::VSELECT : ISD;
- if (!TLI->isOperationExpand(VecISD, LT.second))
- return LT.first;
+ if (!TLI->isOperationExpand(ISD, LT.second) && !LT.second.isVector()) {
+ // The operation is legal. Assume it costs 1. Multiply
+ // by the type-legalization overhead.
+ return LT.first * 1;
}
- // Otherwise, assume vector needs to be scalarized.
- InstructionCost ScalarCost =
- getScalarCost(getTypeLegalizationCost(ValVTy->getScalarType()));
+ // Return the cost of multiple scalar invocations plus the cost of
+ // inserting and extracting the values.
+ auto *ValVTy = cast<FixedVectorType>(ValTy);
+ unsigned Num = ValVTy->getNumElements();
+ InstructionCost ScalarCost = getCmpSelInstrCost(
+ Opcode, ValTy->getScalarType(), CondTy->getScalarType(), VecPred,
+ CostKind, Op1Info, Op2Info, I);
+
+ InstructionCost Overhead =
+ getScalarizationOverhead(ValVTy, /*Insert*/ true,
+ /*Extract*/ true, CostKind);
- // Return scalar cost plus cost of inserting all values.
- return getScalarizationOverhead(ValVTy, /*Insert*/ true,
- /*Extract*/ false, CostKind) +
- ValVTy->getNumElements() * ScalarCost;
+ return Overhead + Num * ScalarCost;
}
InstructionCost
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll
index 57ca3db075689..8947279e9b052 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll
@@ -357,17 +357,20 @@ define <4 x i16> @uadd_sat_v4i16(<4 x i16> %arg0, <4 x i16> %arg1) {
; GFX8-NEXT: bb:
; GFX8-NEXT: [[ARG0_0:%.*]] = extractelement <4 x i16> [[ARG0:%.*]], i64 0
; GFX8-NEXT: [[ARG0_1:%.*]] = extractelement <4 x i16> [[ARG0]], i64 1
+; GFX8-NEXT: [[ARG0_2:%.*]] = extractelement <4 x i16> [[ARG0]], i64 2
+; GFX8-NEXT: [[ARG0_3:%.*]] = extractelement <4 x i16> [[ARG0]], i64 3
; GFX8-NEXT: [[ARG1_0:%.*]] = extractelement <4 x i16> [[ARG1:%.*]], i64 0
; GFX8-NEXT: [[ARG1_1:%.*]] = extractelement <4 x i16> [[ARG1]], i64 1
+; GFX8-NEXT: [[ARG1_2:%.*]] = extractelement <4 x i16> [[ARG1]], i64 2
+; GFX8-NEXT: [[ARG1_3:%.*]] = extractelement <4 x i16> [[ARG1]], i64 3
; GFX8-NEXT: [[ADD_0:%.*]] = call i16 @llvm.umin.i16(i16 [[ARG0_0]], i16 [[ARG1_0]])
; GFX8-NEXT: [[ADD_1:%.*]] = call i16 @llvm.umin.i16(i16 [[ARG0_1]], i16 [[ARG1_1]])
-; GFX8-NEXT: [[TMP0:%.*]] = shufflevector <4 x i16> [[ARG0]], <4 x i16> poison, <2 x i32> <i32 2, i32 3>
-; GFX8-NEXT: [[TMP3:%.*]] = shufflevector <4 x i16> [[ARG1]], <4 x i16> poison, <2 x i32> <i32 2, i32 3>
-; GFX8-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.umin.v2i16(<2 x i16> [[TMP0]], <2 x i16> [[TMP3]])
+; GFX8-NEXT: [[ADD_2:%.*]] = call i16 @llvm.umin.i16(i16 [[ARG0_2]], i16 [[ARG1_2]])
+; GFX8-NEXT: [[ADD_3:%.*]] = call i16 @llvm.umin.i16(i16 [[ARG0_3]], i16 [[ARG1_3]])
; GFX8-NEXT: [[INS_0:%.*]] = insertelement <4 x i16> undef, i16 [[ADD_0]], i64 0
; GFX8-NEXT: [[INS_1:%.*]] = insertelement <4 x i16> [[INS_0]], i16 [[ADD_1]], i64 1
-; GFX8-NEXT: [[TMP2:%.*]] = shufflevector <2 x i16> [[TMP1]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; GFX8-NEXT: [[INS_31:%.*]] = shufflevector <4 x i16> [[INS_1]], <4 x i16> [[TMP2]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; GFX8-NEXT: [[INS_2:%.*]] = insertelement <4 x i16> [[INS_1]], i16 [[ADD_2]], i64 2
+; GFX8-NEXT: [[INS_31:%.*]] = insertelement <4 x i16> [[INS_2]], i16 [[ADD_3]], i64 3
; GFX8-NEXT: ret <4 x i16> [[INS_31]]
;
; GFX9-LABEL: @uadd_sat_v4i16(
>From 61f0f7c8ef85b9e2edffba3365117e17c9e662ca Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Fri, 31 Jul 2026 06:30:51 -0400
Subject: [PATCH 08/20] Use new triple format
---
.../SimplifyCFG/AMDGPU/speculate-block-regpressure.ll | 8 ++++----
.../test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll | 4 ++--
2 files changed, 6 insertions(+), 6 deletions(-)
diff --git a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
index 818fbc7928188..0bb27af23a27a 100644
--- a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
+++ b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
@@ -1,7 +1,7 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
-; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a < %s | FileCheck %s --check-prefix=IR
-; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a < %s \
-; RUN: | llc -O3 -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a | FileCheck %s --check-prefix=ASM
+; RUN: opt -S -passes=simplifycfg -mtriple=amdgpu9.0a-amd-amdhsa < %s | FileCheck %s --check-prefix=IR
+; RUN: opt -S -passes=simplifycfg -mtriple=amdgpu9.0a-amd-amdhsa < %s \
+; RUN: | llc -O3 -mtriple=amdgpu9.0a-amd-amdhsa | FileCheck %s --check-prefix=ASM
; Regression test for SGPR spilling caused by commit 0967957d7a94
; "[CostModel] Handle all cost kinds in getCmpSelInstrCost". That
@@ -14,7 +14,7 @@
define amdgpu_kernel void @reduced_fmha_kernel(i32 %arg, i32 %arg1, i32 %arg2, i1 %arg3, i32 %arg4, i1 %arg5, i1 %arg6, i1 %arg7, i1 %arg8, i32 %arg9, i1 %arg10, <4 x float> %arg11, i32 %arg12, i32 %arg13, i1 %arg14, i1 %arg15, i1 %arg16, i1 %arg17, i32 %arg18, i1 %arg19, i1 %arg20, i32 %arg21, i1 %arg22, i32 %arg23, i1 %arg24, i1 %arg25, i1 %arg26, i32 %arg27, i32 %arg28, i1 %arg29, i32 %arg30, i1 %arg31, i32 %arg32, i1 %arg33, i32 %arg34, <4 x float> %arg35, <4 x float> %arg36, <4 x float> %arg37) {
; IR-LABEL: define amdgpu_kernel void @reduced_fmha_kernel(
-; IR-SAME: i32 [[ARG:%.*]], i32 [[ARG1:%.*]], i32 [[ARG2:%.*]], i1 [[ARG3:%.*]], i32 [[ARG4:%.*]], i1 [[ARG5:%.*]], i1 [[ARG6:%.*]], i1 [[ARG7:%.*]], i1 [[ARG8:%.*]], i32 [[ARG9:%.*]], i1 [[ARG10:%.*]], <4 x float> [[ARG11:%.*]], i32 [[ARG12:%.*]], i32 [[ARG13:%.*]], i1 [[ARG14:%.*]], i1 [[ARG15:%.*]], i1 [[ARG16:%.*]], i1 [[ARG17:%.*]], i32 [[ARG18:%.*]], i1 [[ARG19:%.*]], i1 [[ARG20:%.*]], i32 [[ARG21:%.*]], i1 [[ARG22:%.*]], i32 [[ARG23:%.*]], i1 [[ARG24:%.*]], i1 [[ARG25:%.*]], i1 [[ARG26:%.*]], i32 [[ARG27:%.*]], i32 [[ARG28:%.*]], i1 [[ARG29:%.*]], i32 [[ARG30:%.*]], i1 [[ARG31:%.*]], i32 [[ARG32:%.*]], i1 [[ARG33:%.*]], i32 [[ARG34:%.*]], <4 x float> [[ARG35:%.*]], <4 x float> [[ARG36:%.*]], <4 x float> [[ARG37:%.*]]) #[[ATTR0:[0-9]+]] {
+; IR-SAME: i32 [[ARG:%.*]], i32 [[ARG1:%.*]], i32 [[ARG2:%.*]], i1 [[ARG3:%.*]], i32 [[ARG4:%.*]], i1 [[ARG5:%.*]], i1 [[ARG6:%.*]], i1 [[ARG7:%.*]], i1 [[ARG8:%.*]], i32 [[ARG9:%.*]], i1 [[ARG10:%.*]], <4 x float> [[ARG11:%.*]], i32 [[ARG12:%.*]], i32 [[ARG13:%.*]], i1 [[ARG14:%.*]], i1 [[ARG15:%.*]], i1 [[ARG16:%.*]], i1 [[ARG17:%.*]], i32 [[ARG18:%.*]], i1 [[ARG19:%.*]], i1 [[ARG20:%.*]], i32 [[ARG21:%.*]], i1 [[ARG22:%.*]], i32 [[ARG23:%.*]], i1 [[ARG24:%.*]], i1 [[ARG25:%.*]], i1 [[ARG26:%.*]], i32 [[ARG27:%.*]], i32 [[ARG28:%.*]], i1 [[ARG29:%.*]], i32 [[ARG30:%.*]], i1 [[ARG31:%.*]], i32 [[ARG32:%.*]], i1 [[ARG33:%.*]], i32 [[ARG34:%.*]], <4 x float> [[ARG35:%.*]], <4 x float> [[ARG36:%.*]], <4 x float> [[ARG37:%.*]]) {
; IR-NEXT: [[BB:.*]]:
; IR-NEXT: br label %[[BB38:.*]]
; IR: [[BB38]]:
diff --git a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll
index 37327b63c613a..b1d4295fe7311 100644
--- a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll
+++ b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block.ll
@@ -1,5 +1,5 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
-; RUN: opt -S -passes=simplifycfg -mtriple=amdgcn-amd-amdhsa < %s | FileCheck %s
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=simplifycfg -mtriple=amdgpu9.0a-amd-amdhsa < %s | FileCheck %s
; Test that SimplifyCFG speculatively executes blocks on AMDGPU even for vector
; types. The AMDGPU getCmpSelInstrCost override returns a low cost for selects,
>From 7e25003127dbc33bce9d07cfd9c51ec8bd4d6fcc Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Fri, 31 Jul 2026 06:32:17 -0400
Subject: [PATCH 09/20] Add cost model test for selects
---
llvm/test/Analysis/CostModel/AMDGPU/select.ll | 185 ++++++++++++++++++
1 file changed, 185 insertions(+)
create mode 100644 llvm/test/Analysis/CostModel/AMDGPU/select.ll
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/select.ll b/llvm/test/Analysis/CostModel/AMDGPU/select.ll
new file mode 100644
index 0000000000000..9dafcd4b39429
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/AMDGPU/select.ll
@@ -0,0 +1,185 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa < %s | FileCheck -check-prefixes=ALL %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE %s
+
+define void @select_i32(i1 %cond, i32 %a, i32 %b) {
+; ALL-LABEL: 'select_i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, i32 %a, i32 %b
+ ret void
+}
+
+define void @select_i64(i1 %cond, i64 %a, i64 %b) {
+; ALL-LABEL: 'select_i64'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_i64'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, i64 %a, i64 %b
+ ret void
+}
+
+define void @select_f32(i1 %cond, float %a, float %b) {
+; ALL-LABEL: 'select_f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, float %a, float %b
+ ret void
+}
+
+define void @select_f64(i1 %cond, double %a, double %b) {
+; ALL-LABEL: 'select_f64'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_f64'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, double %a, double %b
+ ret void
+}
+
+define void @select_i16(i1 %cond, i16 %a, i16 %b) {
+; ALL-LABEL: 'select_i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, i16 %a, i16 %b
+ ret void
+}
+
+define void @select_f16(i1 %cond, half %a, half %b) {
+; ALL-LABEL: 'select_f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, half %a, half %b
+ ret void
+}
+
+define void @select_v2i32(i1 %cond, <2 x i32> %a, <2 x i32> %b) {
+; ALL-LABEL: 'select_v2i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
+ ret void
+}
+
+define void @select_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
+; ALL-LABEL: 'select_v4i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+ ret void
+}
+
+define void @select_v2i16(i1 %cond, <2 x i16> %a, <2 x i16> %b) {
+; ALL-LABEL: 'select_v2i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
+ ret void
+}
+
+define void @select_v4i16(i1 %cond, <4 x i16> %a, <4 x i16> %b) {
+; ALL-LABEL: 'select_v4i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
+ ret void
+}
+
+define void @select_v2f32(i1 %cond, <2 x float> %a, <2 x float> %b) {
+; ALL-LABEL: 'select_v2f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
+ ret void
+}
+
+define void @select_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) {
+; ALL-LABEL: 'select_v4f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
+ ret void
+}
+
+define void @select_v2f16(i1 %cond, <2 x half> %a, <2 x half> %b) {
+; ALL-LABEL: 'select_v2f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
+ ret void
+}
+
+define void @select_v4f16(i1 %cond, <4 x half> %a, <4 x half> %b) {
+; ALL-LABEL: 'select_v4f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
+ ret void
+}
>From 38703d783c0e0d0b224b1bbd091ef7a5e4aa03cd Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Fri, 31 Jul 2026 08:17:05 -0400
Subject: [PATCH 10/20] Update test checks using update_test_checks version 6
---
.../SimplifyCFG/AMDGPU/speculate-block-regpressure.ll | 3 ++-
1 file changed, 2 insertions(+), 1 deletion(-)
diff --git a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
index 0bb27af23a27a..e414b3d220864 100644
--- a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
+++ b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
@@ -1,4 +1,5 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+
; RUN: opt -S -passes=simplifycfg -mtriple=amdgpu9.0a-amd-amdhsa < %s | FileCheck %s --check-prefix=IR
; RUN: opt -S -passes=simplifycfg -mtriple=amdgpu9.0a-amd-amdhsa < %s \
; RUN: | llc -O3 -mtriple=amdgpu9.0a-amd-amdhsa | FileCheck %s --check-prefix=ASM
>From c8e98b5068b4f130f4c4a562a5531f926e00ffd3 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Mon, 3 Aug 2026 04:56:33 -0400
Subject: [PATCH 11/20] Review changes
- Simplify getCmpSelInstrCost to call base implementation
- Add compare tests to select.ll and rename file
- Add GFX8, GFX9, GFX11 run lines
- Use new amdgpu triple format
- Simplify llc invocation
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 39 +-
.../CostModel/AMDGPU/compare-select.ll | 390 ++++++++++++++++++
llvm/test/Analysis/CostModel/AMDGPU/select.ll | 185 ---------
.../SLPVectorizer/AMDGPU/min_max.ll | 13 +-
.../AMDGPU/speculate-block-regpressure.ll | 2 +-
5 files changed, 398 insertions(+), 231 deletions(-)
create mode 100644 llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
delete mode 100644 llvm/test/Analysis/CostModel/AMDGPU/select.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 727926046f0fe..0c26023cf55f9 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -24,14 +24,11 @@
#include "llvm/Analysis/LoopInfo.h"
#include "llvm/Analysis/ValueTracking.h"
#include "llvm/CodeGen/Analysis.h"
-#include "llvm/IR/DerivedTypes.h"
#include "llvm/IR/Function.h"
#include "llvm/IR/IRBuilder.h"
#include "llvm/IR/IntrinsicsAMDGPU.h"
#include "llvm/IR/PatternMatch.h"
-#include "llvm/Support/Casting.h"
#include "llvm/Support/KnownBits.h"
-#include <llvm/CodeGen/ISDOpcodes.h>
#include <optional>
using namespace llvm;
@@ -987,45 +984,13 @@ InstructionCost GCNTTIImpl::getCmpSelInstrCost(
unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info,
TTI::OperandValueInfo Op2Info, const Instruction *I) const {
- if (isa<ScalableVectorType>(ValTy))
- return InstructionCost::getInvalid();
-
// For size and latency cost kinds, return a low cost independent of vector
// width to enable SimplifyCFG's speculativelyExecuteBB optimization.
if (CostKind != TTI::TCK_RecipThroughput)
return 1;
- // Compute cost based on type legalization.
- const TargetLoweringBase *TLI = getTLI();
- if (TLI->getValueType(DL, ValTy, true) == MVT::Other)
- return 1;
-
- const int ISD = TLI->InstructionOpcodeToISD(Opcode);
- assert(ISD && "Invalid opcode");
- std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
-
- if (!ValTy->isVectorTy())
- return TLI->isOperationExpand(ISD, LT.second) ? 1 : LT.first;
-
- if (!TLI->isOperationExpand(ISD, LT.second) && !LT.second.isVector()) {
- // The operation is legal. Assume it costs 1. Multiply
- // by the type-legalization overhead.
- return LT.first * 1;
- }
-
- // Return the cost of multiple scalar invocations plus the cost of
- // inserting and extracting the values.
- auto *ValVTy = cast<FixedVectorType>(ValTy);
- unsigned Num = ValVTy->getNumElements();
- InstructionCost ScalarCost = getCmpSelInstrCost(
- Opcode, ValTy->getScalarType(), CondTy->getScalarType(), VecPred,
- CostKind, Op1Info, Op2Info, I);
-
- InstructionCost Overhead =
- getScalarizationOverhead(ValVTy, /*Insert*/ true,
- /*Extract*/ true, CostKind);
-
- return Overhead + Num * ScalarCost;
+ return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
+ Op1Info, Op2Info, I);
}
InstructionCost
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll b/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
new file mode 100644
index 0000000000000..e004707914511
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
@@ -0,0 +1,390 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu8.03-amd-amdhsa < %s | FileCheck -check-prefixes=ALL,GFX8 %s
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL,GFX9 %s
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu11.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL,GFX11 %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE %s
+
+define void @icmp_i32(i32 %a, i32 %b) {
+; ALL-LABEL: 'icmp_i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i32 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_ne = icmp ne i32 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i32 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_sge = icmp sge i32 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'icmp_i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i32 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_ne = icmp ne i32 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i32 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_sge = icmp sge i32 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp_eq = icmp eq i32 %a, %b
+ %cmp_ne = icmp ne i32 %a, %b
+ %cmp_slt = icmp slt i32 %a, %b
+ %cmp_sge = icmp sge i32 %a, %b
+ ret void
+}
+
+define void @icmp_i64(i64 %a, i64 %b) {
+; ALL-LABEL: 'icmp_i64'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i64 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i64 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'icmp_i64'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i64 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i64 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp_eq = icmp eq i64 %a, %b
+ %cmp_slt = icmp slt i64 %a, %b
+ ret void
+}
+
+define void @icmp_i16(i16 %a, i16 %b) {
+; ALL-LABEL: 'icmp_i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i16 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i16 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'icmp_i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i16 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i16 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp_eq = icmp eq i16 %a, %b
+ %cmp_slt = icmp slt i16 %a, %b
+ ret void
+}
+
+define void @fcmp_f32(float %a, float %b) {
+; ALL-LABEL: 'fcmp_f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq float %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_one = fcmp one float %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt float %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oge = fcmp oge float %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'fcmp_f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq float %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_one = fcmp one float %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt float %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oge = fcmp oge float %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp_oeq = fcmp oeq float %a, %b
+ %cmp_one = fcmp one float %a, %b
+ %cmp_olt = fcmp olt float %a, %b
+ %cmp_oge = fcmp oge float %a, %b
+ ret void
+}
+
+define void @fcmp_f64(double %a, double %b) {
+; ALL-LABEL: 'fcmp_f64'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq double %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt double %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'fcmp_f64'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq double %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt double %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp_oeq = fcmp oeq double %a, %b
+ %cmp_olt = fcmp olt double %a, %b
+ ret void
+}
+
+define void @fcmp_f16(half %a, half %b) {
+; ALL-LABEL: 'fcmp_f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq half %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt half %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'fcmp_f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq half %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt half %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp_oeq = fcmp oeq half %a, %b
+ %cmp_olt = fcmp olt half %a, %b
+ ret void
+}
+
+define void @select_i32(i1 %cond, i32 %a, i32 %b) {
+; ALL-LABEL: 'select_i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, i32 %a, i32 %b
+ ret void
+}
+
+define void @select_i64(i1 %cond, i64 %a, i64 %b) {
+; ALL-LABEL: 'select_i64'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_i64'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, i64 %a, i64 %b
+ ret void
+}
+
+define void @select_f32(i1 %cond, float %a, float %b) {
+; ALL-LABEL: 'select_f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, float %a, float %b
+ ret void
+}
+
+define void @select_f64(i1 %cond, double %a, double %b) {
+; ALL-LABEL: 'select_f64'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_f64'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, double %a, double %b
+ ret void
+}
+
+define void @select_i16(i1 %cond, i16 %a, i16 %b) {
+; ALL-LABEL: 'select_i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, i16 %a, i16 %b
+ ret void
+}
+
+define void @select_f16(i1 %cond, half %a, half %b) {
+; ALL-LABEL: 'select_f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, half %a, half %b
+ ret void
+}
+
+define void @select_ptr(i1 %cond, ptr %a, ptr %b) {
+; ALL-LABEL: 'select_ptr'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, ptr %a, ptr %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_ptr'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, ptr %a, ptr %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, ptr %a, ptr %b
+ ret void
+}
+
+define void @select_v2i32(i1 %cond, <2 x i32> %a, <2 x i32> %b) {
+; ALL-LABEL: 'select_v2i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
+ ret void
+}
+
+define void @select_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
+; ALL-LABEL: 'select_v4i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+ ret void
+}
+
+define void @select_v2i16(i1 %cond, <2 x i16> %a, <2 x i16> %b) {
+; ALL-LABEL: 'select_v2i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
+ ret void
+}
+
+define void @select_v4i16(i1 %cond, <4 x i16> %a, <4 x i16> %b) {
+; ALL-LABEL: 'select_v4i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
+ ret void
+}
+
+define void @select_v2f32(i1 %cond, <2 x float> %a, <2 x float> %b) {
+; ALL-LABEL: 'select_v2f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
+ ret void
+}
+
+define void @select_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) {
+; ALL-LABEL: 'select_v4f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
+ ret void
+}
+
+define void @select_v2f16(i1 %cond, <2 x half> %a, <2 x half> %b) {
+; ALL-LABEL: 'select_v2f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
+ ret void
+}
+
+define void @select_v4f16(i1 %cond, <4 x half> %a, <4 x half> %b) {
+; ALL-LABEL: 'select_v4f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
+ ret void
+}
+
+define void @select_v3i32(i1 %cond, <3 x i32> %a, <3 x i32> %b) {
+; ALL-LABEL: 'select_v3i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v3i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
+ ret void
+}
+
+define void @select_v3f32(i1 %cond, <3 x float> %a, <3 x float> %b) {
+; ALL-LABEL: 'select_v3f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v3f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
+ ret void
+}
+
+define void @select_v3i16(i1 %cond, <3 x i16> %a, <3 x i16> %b) {
+; ALL-LABEL: 'select_v3i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v3i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
+ ret void
+}
+
+define void @select_v3f16(i1 %cond, <3 x half> %a, <3 x half> %b) {
+; ALL-LABEL: 'select_v3f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v3f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
+ ret void
+}
+
+define void @select_v2ptr(i1 %cond, <2 x ptr> %a, <2 x ptr> %b) {
+; ALL-LABEL: 'select_v2ptr'
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2ptr'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
+ ret void
+}
+
+define void @select_v4ptr(i1 %cond, <4 x ptr> %a, <4 x ptr> %b) {
+; ALL-LABEL: 'select_v4ptr'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4ptr'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
+ ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX11: {{.*}}
+; GFX8: {{.*}}
+; GFX9: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/select.ll b/llvm/test/Analysis/CostModel/AMDGPU/select.ll
deleted file mode 100644
index d94fb79cb6f9e..0000000000000
--- a/llvm/test/Analysis/CostModel/AMDGPU/select.ll
+++ /dev/null
@@ -1,185 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa < %s | FileCheck -check-prefixes=ALL %s
-; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgcn-unknown-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE %s
-
-define void @select_i32(i1 %cond, i32 %a, i32 %b) {
-; ALL-LABEL: 'select_i32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_i32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, i32 %a, i32 %b
- ret void
-}
-
-define void @select_i64(i1 %cond, i64 %a, i64 %b) {
-; ALL-LABEL: 'select_i64'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_i64'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, i64 %a, i64 %b
- ret void
-}
-
-define void @select_f32(i1 %cond, float %a, float %b) {
-; ALL-LABEL: 'select_f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_f32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, float %a, float %b
- ret void
-}
-
-define void @select_f64(i1 %cond, double %a, double %b) {
-; ALL-LABEL: 'select_f64'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_f64'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, double %a, double %b
- ret void
-}
-
-define void @select_i16(i1 %cond, i16 %a, i16 %b) {
-; ALL-LABEL: 'select_i16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_i16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, i16 %a, i16 %b
- ret void
-}
-
-define void @select_f16(i1 %cond, half %a, half %b) {
-; ALL-LABEL: 'select_f16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_f16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, half %a, half %b
- ret void
-}
-
-define void @select_v2i32(i1 %cond, <2 x i32> %a, <2 x i32> %b) {
-; ALL-LABEL: 'select_v2i32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v2i32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
- ret void
-}
-
-define void @select_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
-; ALL-LABEL: 'select_v4i32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v4i32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
- ret void
-}
-
-define void @select_v2i16(i1 %cond, <2 x i16> %a, <2 x i16> %b) {
-; ALL-LABEL: 'select_v2i16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v2i16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
- ret void
-}
-
-define void @select_v4i16(i1 %cond, <4 x i16> %a, <4 x i16> %b) {
-; ALL-LABEL: 'select_v4i16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v4i16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
- ret void
-}
-
-define void @select_v2f32(i1 %cond, <2 x float> %a, <2 x float> %b) {
-; ALL-LABEL: 'select_v2f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v2f32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
- ret void
-}
-
-define void @select_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) {
-; ALL-LABEL: 'select_v4f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v4f32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
- ret void
-}
-
-define void @select_v2f16(i1 %cond, <2 x half> %a, <2 x half> %b) {
-; ALL-LABEL: 'select_v2f16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v2f16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
- ret void
-}
-
-define void @select_v4f16(i1 %cond, <4 x half> %a, <4 x half> %b) {
-; ALL-LABEL: 'select_v4f16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v4f16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
- ret void
-}
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll
index 4b3e8f1487186..64ecaca17380d 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll
@@ -357,20 +357,17 @@ define <4 x i16> @uadd_sat_v4i16(<4 x i16> %arg0, <4 x i16> %arg1) {
; GFX8-NEXT: bb:
; GFX8-NEXT: [[ARG0_0:%.*]] = extractelement <4 x i16> [[ARG0:%.*]], i64 0
; GFX8-NEXT: [[ARG0_1:%.*]] = extractelement <4 x i16> [[ARG0]], i64 1
-; GFX8-NEXT: [[ARG0_2:%.*]] = extractelement <4 x i16> [[ARG0]], i64 2
-; GFX8-NEXT: [[ARG0_3:%.*]] = extractelement <4 x i16> [[ARG0]], i64 3
; GFX8-NEXT: [[ARG1_0:%.*]] = extractelement <4 x i16> [[ARG1:%.*]], i64 0
; GFX8-NEXT: [[ARG1_1:%.*]] = extractelement <4 x i16> [[ARG1]], i64 1
-; GFX8-NEXT: [[ARG1_2:%.*]] = extractelement <4 x i16> [[ARG1]], i64 2
-; GFX8-NEXT: [[ARG1_3:%.*]] = extractelement <4 x i16> [[ARG1]], i64 3
; GFX8-NEXT: [[ADD_0:%.*]] = call i16 @llvm.umin.i16(i16 [[ARG0_0]], i16 [[ARG1_0]])
; GFX8-NEXT: [[ADD_1:%.*]] = call i16 @llvm.umin.i16(i16 [[ARG0_1]], i16 [[ARG1_1]])
-; GFX8-NEXT: [[ADD_2:%.*]] = call i16 @llvm.umin.i16(i16 [[ARG0_2]], i16 [[ARG1_2]])
-; GFX8-NEXT: [[ADD_3:%.*]] = call i16 @llvm.umin.i16(i16 [[ARG0_3]], i16 [[ARG1_3]])
+; GFX8-NEXT: [[TMP0:%.*]] = shufflevector <4 x i16> [[ARG0]], <4 x i16> poison, <2 x i32> <i32 2, i32 3>
+; GFX8-NEXT: [[TMP1:%.*]] = shufflevector <4 x i16> [[ARG1]], <4 x i16> poison, <2 x i32> <i32 2, i32 3>
+; GFX8-NEXT: [[TMP2:%.*]] = call <2 x i16> @llvm.umin.v2i16(<2 x i16> [[TMP0]], <2 x i16> [[TMP1]])
; GFX8-NEXT: [[INS_0:%.*]] = insertelement <4 x i16> undef, i16 [[ADD_0]], i64 0
; GFX8-NEXT: [[INS_1:%.*]] = insertelement <4 x i16> [[INS_0]], i16 [[ADD_1]], i64 1
-; GFX8-NEXT: [[INS_2:%.*]] = insertelement <4 x i16> [[INS_1]], i16 [[ADD_2]], i64 2
-; GFX8-NEXT: [[INS_31:%.*]] = insertelement <4 x i16> [[INS_2]], i16 [[ADD_3]], i64 3
+; GFX8-NEXT: [[TMP3:%.*]] = shufflevector <2 x i16> [[TMP2]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; GFX8-NEXT: [[INS_31:%.*]] = shufflevector <4 x i16> [[INS_1]], <4 x i16> [[TMP3]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
; GFX8-NEXT: ret <4 x i16> [[INS_31]]
;
; GFX9-LABEL: @uadd_sat_v4i16(
diff --git a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
index e414b3d220864..541f205435cb4 100644
--- a/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
+++ b/llvm/test/Transforms/SimplifyCFG/AMDGPU/speculate-block-regpressure.ll
@@ -2,7 +2,7 @@
; RUN: opt -S -passes=simplifycfg -mtriple=amdgpu9.0a-amd-amdhsa < %s | FileCheck %s --check-prefix=IR
; RUN: opt -S -passes=simplifycfg -mtriple=amdgpu9.0a-amd-amdhsa < %s \
-; RUN: | llc -O3 -mtriple=amdgpu9.0a-amd-amdhsa | FileCheck %s --check-prefix=ASM
+; RUN: | llc | FileCheck %s --check-prefix=ASM
; Regression test for SGPR spilling caused by commit 0967957d7a94
; "[CostModel] Handle all cost kinds in getCmpSelInstrCost". That
>From 2ae99a056e4e545836192bca000f3010dc44155e Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Wed, 9 Sep 2026 08:16:24 -0400
Subject: [PATCH 12/20] WIP
---
.../CostModel/AMDGPU/compare-select.ll | 119 ++++++++++++++++++
.../Analysis/CostModel/AMDGPU/reduce-and.ll | 32 ++---
.../Analysis/CostModel/AMDGPU/reduce-or.ll | 32 ++---
3 files changed, 151 insertions(+), 32 deletions(-)
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll b/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
index e004707914511..8b01d530436da 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
@@ -3,6 +3,7 @@
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL,GFX9 %s
; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu11.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL,GFX11 %s
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=size-latency 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE-LATENCY %s
define void @icmp_i32(i32 %a, i32 %b) {
; ALL-LABEL: 'icmp_i32'
@@ -18,6 +19,13 @@ define void @icmp_i32(i32 %a, i32 %b) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i32 %a, %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_sge = icmp sge i32 %a, %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'icmp_i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i32 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_ne = icmp ne i32 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i32 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_sge = icmp sge i32 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp_eq = icmp eq i32 %a, %b
%cmp_ne = icmp ne i32 %a, %b
@@ -36,6 +44,11 @@ define void @icmp_i64(i64 %a, i64 %b) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i64 %a, %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i64 %a, %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'icmp_i64'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i64 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i64 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp_eq = icmp eq i64 %a, %b
%cmp_slt = icmp slt i64 %a, %b
@@ -52,6 +65,11 @@ define void @icmp_i16(i16 %a, i16 %b) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i16 %a, %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i16 %a, %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'icmp_i16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i16 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i16 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp_eq = icmp eq i16 %a, %b
%cmp_slt = icmp slt i16 %a, %b
@@ -72,6 +90,13 @@ define void @fcmp_f32(float %a, float %b) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt float %a, %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oge = fcmp oge float %a, %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'fcmp_f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq float %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_one = fcmp one float %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt float %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oge = fcmp oge float %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp_oeq = fcmp oeq float %a, %b
%cmp_one = fcmp one float %a, %b
@@ -90,6 +115,11 @@ define void @fcmp_f64(double %a, double %b) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq double %a, %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt double %a, %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'fcmp_f64'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq double %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt double %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp_oeq = fcmp oeq double %a, %b
%cmp_olt = fcmp olt double %a, %b
@@ -106,6 +136,11 @@ define void @fcmp_f16(half %a, half %b) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq half %a, %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt half %a, %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'fcmp_f16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq half %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt half %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp_oeq = fcmp oeq half %a, %b
%cmp_olt = fcmp olt half %a, %b
@@ -120,6 +155,10 @@ define void @select_i32(i1 %cond, i32 %a, i32 %b) {
; ALL-SIZE-LABEL: 'select_i32'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, i32 %a, i32 %b
ret void
@@ -133,6 +172,10 @@ define void @select_i64(i1 %cond, i64 %a, i64 %b) {
; ALL-SIZE-LABEL: 'select_i64'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_i64'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, i64 %a, i64 %b
ret void
@@ -146,6 +189,10 @@ define void @select_f32(i1 %cond, float %a, float %b) {
; ALL-SIZE-LABEL: 'select_f32'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, float %a, float %b
ret void
@@ -159,6 +206,10 @@ define void @select_f64(i1 %cond, double %a, double %b) {
; ALL-SIZE-LABEL: 'select_f64'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_f64'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, double %a, double %b
ret void
@@ -172,6 +223,10 @@ define void @select_i16(i1 %cond, i16 %a, i16 %b) {
; ALL-SIZE-LABEL: 'select_i16'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_i16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, i16 %a, i16 %b
ret void
@@ -185,6 +240,10 @@ define void @select_f16(i1 %cond, half %a, half %b) {
; ALL-SIZE-LABEL: 'select_f16'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_f16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, half %a, half %b
ret void
@@ -198,6 +257,10 @@ define void @select_ptr(i1 %cond, ptr %a, ptr %b) {
; ALL-SIZE-LABEL: 'select_ptr'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, ptr %a, ptr %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_ptr'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, ptr %a, ptr %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, ptr %a, ptr %b
ret void
@@ -211,6 +274,10 @@ define void @select_v2i32(i1 %cond, <2 x i32> %a, <2 x i32> %b) {
; ALL-SIZE-LABEL: 'select_v2i32'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v2i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
ret void
@@ -224,6 +291,10 @@ define void @select_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
; ALL-SIZE-LABEL: 'select_v4i32'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v4i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
ret void
@@ -237,6 +308,10 @@ define void @select_v2i16(i1 %cond, <2 x i16> %a, <2 x i16> %b) {
; ALL-SIZE-LABEL: 'select_v2i16'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v2i16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
ret void
@@ -250,6 +325,10 @@ define void @select_v4i16(i1 %cond, <4 x i16> %a, <4 x i16> %b) {
; ALL-SIZE-LABEL: 'select_v4i16'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v4i16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
ret void
@@ -263,6 +342,10 @@ define void @select_v2f32(i1 %cond, <2 x float> %a, <2 x float> %b) {
; ALL-SIZE-LABEL: 'select_v2f32'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v2f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <2 x float> %a, <2 x float> %b
ret void
@@ -276,6 +359,10 @@ define void @select_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) {
; ALL-SIZE-LABEL: 'select_v4f32'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v4f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <4 x float> %a, <4 x float> %b
ret void
@@ -289,6 +376,10 @@ define void @select_v2f16(i1 %cond, <2 x half> %a, <2 x half> %b) {
; ALL-SIZE-LABEL: 'select_v2f16'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v2f16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <2 x half> %a, <2 x half> %b
ret void
@@ -302,6 +393,10 @@ define void @select_v4f16(i1 %cond, <4 x half> %a, <4 x half> %b) {
; ALL-SIZE-LABEL: 'select_v4f16'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v4f16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <4 x half> %a, <4 x half> %b
ret void
@@ -315,6 +410,10 @@ define void @select_v3i32(i1 %cond, <3 x i32> %a, <3 x i32> %b) {
; ALL-SIZE-LABEL: 'select_v3i32'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v3i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
ret void
@@ -328,6 +427,10 @@ define void @select_v3f32(i1 %cond, <3 x float> %a, <3 x float> %b) {
; ALL-SIZE-LABEL: 'select_v3f32'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v3f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <3 x float> %a, <3 x float> %b
ret void
@@ -341,6 +444,10 @@ define void @select_v3i16(i1 %cond, <3 x i16> %a, <3 x i16> %b) {
; ALL-SIZE-LABEL: 'select_v3i16'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v3i16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
ret void
@@ -354,6 +461,10 @@ define void @select_v3f16(i1 %cond, <3 x half> %a, <3 x half> %b) {
; ALL-SIZE-LABEL: 'select_v3f16'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v3f16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <3 x half> %a, <3 x half> %b
ret void
@@ -367,6 +478,10 @@ define void @select_v2ptr(i1 %cond, <2 x ptr> %a, <2 x ptr> %b) {
; ALL-SIZE-LABEL: 'select_v2ptr'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v2ptr'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
ret void
@@ -380,6 +495,10 @@ define void @select_v4ptr(i1 %cond, <4 x ptr> %a, <4 x ptr> %b) {
; ALL-SIZE-LABEL: 'select_v4ptr'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v4ptr'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
ret void
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll b/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll
index fa7c6410bc719..b0b0b33c947b4 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll
@@ -6,26 +6,26 @@
define i32 @reduce_i1(i32 %arg) {
; ALL-LABEL: 'reduce_i1'
; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %V1 = call i1 @llvm.vector.reduce.and.v1i1(<1 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %V2 = call i1 @llvm.vector.reduce.and.v2i1(<2 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 5 for instruction: %V4 = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %V8 = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V16 = call i1 @llvm.vector.reduce.and.v16i1(<16 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V32 = call i1 @llvm.vector.reduce.and.v32i1(<32 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V64 = call i1 @llvm.vector.reduce.and.v64i1(<64 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 130 for instruction: %V128 = call i1 @llvm.vector.reduce.and.v128i1(<128 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 260 for instruction: %V256 = call i1 @llvm.vector.reduce.and.v256i1(<256 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %V2 = call i1 @llvm.vector.reduce.and.v2i1(<2 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V4 = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V8 = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V16 = call i1 @llvm.vector.reduce.and.v16i1(<16 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V32 = call i1 @llvm.vector.reduce.and.v32i1(<32 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 257 for instruction: %V64 = call i1 @llvm.vector.reduce.and.v64i1(<64 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 514 for instruction: %V128 = call i1 @llvm.vector.reduce.and.v128i1(<128 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 1028 for instruction: %V256 = call i1 @llvm.vector.reduce.and.v256i1(<256 x i1> undef)
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
; ALL-SIZE-LABEL: 'reduce_i1'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %V1 = call i1 @llvm.vector.reduce.and.v1i1(<1 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %V2 = call i1 @llvm.vector.reduce.and.v2i1(<2 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 5 for instruction: %V4 = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %V8 = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V16 = call i1 @llvm.vector.reduce.and.v16i1(<16 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V32 = call i1 @llvm.vector.reduce.and.v32i1(<32 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V64 = call i1 @llvm.vector.reduce.and.v64i1(<64 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V128 = call i1 @llvm.vector.reduce.and.v128i1(<128 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 257 for instruction: %V256 = call i1 @llvm.vector.reduce.and.v256i1(<256 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %V2 = call i1 @llvm.vector.reduce.and.v2i1(<2 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V4 = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V8 = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V16 = call i1 @llvm.vector.reduce.and.v16i1(<16 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V32 = call i1 @llvm.vector.reduce.and.v32i1(<32 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 257 for instruction: %V64 = call i1 @llvm.vector.reduce.and.v64i1(<64 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 513 for instruction: %V128 = call i1 @llvm.vector.reduce.and.v128i1(<128 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1025 for instruction: %V256 = call i1 @llvm.vector.reduce.and.v256i1(<256 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
;
%V1 = call i1 @llvm.vector.reduce.and.v1i1(<1 x i1> undef)
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll b/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll
index 70367da9db431..831499b4aece7 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll
@@ -6,26 +6,26 @@
define i32 @reduce_i1(i32 %arg) {
; ALL-LABEL: 'reduce_i1'
; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %V1 = call i1 @llvm.vector.reduce.or.v1i1(<1 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %V2 = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 5 for instruction: %V4 = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %V8 = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V16 = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V32 = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V64 = call i1 @llvm.vector.reduce.or.v64i1(<64 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 130 for instruction: %V128 = call i1 @llvm.vector.reduce.or.v128i1(<128 x i1> undef)
-; ALL-NEXT: Cost Model: Found an estimated cost of 260 for instruction: %V256 = call i1 @llvm.vector.reduce.or.v256i1(<256 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %V2 = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V4 = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V8 = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V16 = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V32 = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 257 for instruction: %V64 = call i1 @llvm.vector.reduce.or.v64i1(<64 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 514 for instruction: %V128 = call i1 @llvm.vector.reduce.or.v128i1(<128 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 1028 for instruction: %V256 = call i1 @llvm.vector.reduce.or.v256i1(<256 x i1> undef)
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret i32 undef
;
; ALL-SIZE-LABEL: 'reduce_i1'
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %V1 = call i1 @llvm.vector.reduce.or.v1i1(<1 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %V2 = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 5 for instruction: %V4 = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %V8 = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V16 = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V32 = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V64 = call i1 @llvm.vector.reduce.or.v64i1(<64 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V128 = call i1 @llvm.vector.reduce.or.v128i1(<128 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 257 for instruction: %V256 = call i1 @llvm.vector.reduce.or.v256i1(<256 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %V2 = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V4 = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V8 = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V16 = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V32 = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 257 for instruction: %V64 = call i1 @llvm.vector.reduce.or.v64i1(<64 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 513 for instruction: %V128 = call i1 @llvm.vector.reduce.or.v128i1(<128 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1025 for instruction: %V256 = call i1 @llvm.vector.reduce.or.v256i1(<256 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
;
%V1 = call i1 @llvm.vector.reduce.or.v1i1(<1 x i1> undef)
>From 73d3e040df1e68a01e81ac7a835be334987f80ee Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Wed, 9 Sep 2026 09:32:12 -0400
Subject: [PATCH 13/20] Add unroll test
Demonstrates negative effect of the compare/select cost change on unrolling.
---
.../AMDGPU/unroll-cost-cmp-select.ll | 157 ++++++++++++++++++
1 file changed, 157 insertions(+)
create mode 100644 llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll
new file mode 100644
index 0000000000000..87b47ad389f07
--- /dev/null
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll
@@ -0,0 +1,157 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -mtriple=amdgpu9.0a-amd-amdhsa -passes=loop-unroll < %s | FileCheck %s
+
+; The TCK_CodeSize cost kind used for pricing cmp/selects in loop unrolling
+; should take the vector-width into account which prevents this loop
+; from being unrolled.
+
+define void @wide_select_loop(ptr addrspace(1) %p, <16 x float> %a, <16 x float> %b) {
+; CHECK-LABEL: define void @wide_select_loop(
+; CHECK-SAME: ptr addrspace(1) [[P:%.*]], <16 x float> [[A:%.*]], <16 x float> [[B:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[I:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[INC_7:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[I]]
+; CHECK-NEXT: [[V:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP]], align 64
+; CHECK-NEXT: [[C1:%.*]] = fcmp ogt <16 x float> [[V]], [[A]]
+; CHECK-NEXT: [[S1:%.*]] = select <16 x i1> [[C1]], <16 x float> [[V]], <16 x float> [[B]]
+; CHECK-NEXT: [[C2:%.*]] = fcmp ogt <16 x float> [[S1]], [[A]]
+; CHECK-NEXT: [[S2:%.*]] = select <16 x i1> [[C2]], <16 x float> [[S1]], <16 x float> [[B]]
+; CHECK-NEXT: [[C3:%.*]] = fcmp ogt <16 x float> [[S2]], [[A]]
+; CHECK-NEXT: [[S3:%.*]] = select <16 x i1> [[C3]], <16 x float> [[S2]], <16 x float> [[B]]
+; CHECK-NEXT: [[C4:%.*]] = fcmp ogt <16 x float> [[S3]], [[A]]
+; CHECK-NEXT: [[S4:%.*]] = select <16 x i1> [[C4]], <16 x float> [[S3]], <16 x float> [[B]]
+; CHECK-NEXT: [[C5:%.*]] = fcmp ogt <16 x float> [[S4]], [[A]]
+; CHECK-NEXT: [[S5:%.*]] = select <16 x i1> [[C5]], <16 x float> [[S4]], <16 x float> [[B]]
+; CHECK-NEXT: store <16 x float> [[S5]], ptr addrspace(1) [[GEP]], align 64
+; CHECK-NEXT: [[INC:%.*]] = add nuw nsw i32 [[I]], 1
+; CHECK-NEXT: [[GEP_1:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC]]
+; CHECK-NEXT: [[V_1:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_1]], align 64
+; CHECK-NEXT: [[C1_1:%.*]] = fcmp ogt <16 x float> [[V_1]], [[A]]
+; CHECK-NEXT: [[S1_1:%.*]] = select <16 x i1> [[C1_1]], <16 x float> [[V_1]], <16 x float> [[B]]
+; CHECK-NEXT: [[C2_1:%.*]] = fcmp ogt <16 x float> [[S1_1]], [[A]]
+; CHECK-NEXT: [[S2_1:%.*]] = select <16 x i1> [[C2_1]], <16 x float> [[S1_1]], <16 x float> [[B]]
+; CHECK-NEXT: [[C3_1:%.*]] = fcmp ogt <16 x float> [[S2_1]], [[A]]
+; CHECK-NEXT: [[S3_1:%.*]] = select <16 x i1> [[C3_1]], <16 x float> [[S2_1]], <16 x float> [[B]]
+; CHECK-NEXT: [[C4_1:%.*]] = fcmp ogt <16 x float> [[S3_1]], [[A]]
+; CHECK-NEXT: [[S4_1:%.*]] = select <16 x i1> [[C4_1]], <16 x float> [[S3_1]], <16 x float> [[B]]
+; CHECK-NEXT: [[C5_1:%.*]] = fcmp ogt <16 x float> [[S4_1]], [[A]]
+; CHECK-NEXT: [[S5_1:%.*]] = select <16 x i1> [[C5_1]], <16 x float> [[S4_1]], <16 x float> [[B]]
+; CHECK-NEXT: store <16 x float> [[S5_1]], ptr addrspace(1) [[GEP_1]], align 64
+; CHECK-NEXT: [[INC_1:%.*]] = add nuw nsw i32 [[I]], 2
+; CHECK-NEXT: [[GEP_2:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC_1]]
+; CHECK-NEXT: [[V_2:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_2]], align 64
+; CHECK-NEXT: [[C1_2:%.*]] = fcmp ogt <16 x float> [[V_2]], [[A]]
+; CHECK-NEXT: [[S1_2:%.*]] = select <16 x i1> [[C1_2]], <16 x float> [[V_2]], <16 x float> [[B]]
+; CHECK-NEXT: [[C2_2:%.*]] = fcmp ogt <16 x float> [[S1_2]], [[A]]
+; CHECK-NEXT: [[S2_2:%.*]] = select <16 x i1> [[C2_2]], <16 x float> [[S1_2]], <16 x float> [[B]]
+; CHECK-NEXT: [[C3_2:%.*]] = fcmp ogt <16 x float> [[S2_2]], [[A]]
+; CHECK-NEXT: [[S3_2:%.*]] = select <16 x i1> [[C3_2]], <16 x float> [[S2_2]], <16 x float> [[B]]
+; CHECK-NEXT: [[C4_2:%.*]] = fcmp ogt <16 x float> [[S3_2]], [[A]]
+; CHECK-NEXT: [[S4_2:%.*]] = select <16 x i1> [[C4_2]], <16 x float> [[S3_2]], <16 x float> [[B]]
+; CHECK-NEXT: [[C5_2:%.*]] = fcmp ogt <16 x float> [[S4_2]], [[A]]
+; CHECK-NEXT: [[S5_2:%.*]] = select <16 x i1> [[C5_2]], <16 x float> [[S4_2]], <16 x float> [[B]]
+; CHECK-NEXT: store <16 x float> [[S5_2]], ptr addrspace(1) [[GEP_2]], align 64
+; CHECK-NEXT: [[INC_2:%.*]] = add nuw nsw i32 [[I]], 3
+; CHECK-NEXT: [[GEP_3:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC_2]]
+; CHECK-NEXT: [[V_3:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_3]], align 64
+; CHECK-NEXT: [[C1_3:%.*]] = fcmp ogt <16 x float> [[V_3]], [[A]]
+; CHECK-NEXT: [[S1_3:%.*]] = select <16 x i1> [[C1_3]], <16 x float> [[V_3]], <16 x float> [[B]]
+; CHECK-NEXT: [[C2_3:%.*]] = fcmp ogt <16 x float> [[S1_3]], [[A]]
+; CHECK-NEXT: [[S2_3:%.*]] = select <16 x i1> [[C2_3]], <16 x float> [[S1_3]], <16 x float> [[B]]
+; CHECK-NEXT: [[C3_3:%.*]] = fcmp ogt <16 x float> [[S2_3]], [[A]]
+; CHECK-NEXT: [[S3_3:%.*]] = select <16 x i1> [[C3_3]], <16 x float> [[S2_3]], <16 x float> [[B]]
+; CHECK-NEXT: [[C4_3:%.*]] = fcmp ogt <16 x float> [[S3_3]], [[A]]
+; CHECK-NEXT: [[S4_3:%.*]] = select <16 x i1> [[C4_3]], <16 x float> [[S3_3]], <16 x float> [[B]]
+; CHECK-NEXT: [[C5_3:%.*]] = fcmp ogt <16 x float> [[S4_3]], [[A]]
+; CHECK-NEXT: [[S5_3:%.*]] = select <16 x i1> [[C5_3]], <16 x float> [[S4_3]], <16 x float> [[B]]
+; CHECK-NEXT: store <16 x float> [[S5_3]], ptr addrspace(1) [[GEP_3]], align 64
+; CHECK-NEXT: [[INC_3:%.*]] = add nuw nsw i32 [[I]], 4
+; CHECK-NEXT: [[GEP_4:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC_3]]
+; CHECK-NEXT: [[V_4:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_4]], align 64
+; CHECK-NEXT: [[C1_4:%.*]] = fcmp ogt <16 x float> [[V_4]], [[A]]
+; CHECK-NEXT: [[S1_4:%.*]] = select <16 x i1> [[C1_4]], <16 x float> [[V_4]], <16 x float> [[B]]
+; CHECK-NEXT: [[C2_4:%.*]] = fcmp ogt <16 x float> [[S1_4]], [[A]]
+; CHECK-NEXT: [[S2_4:%.*]] = select <16 x i1> [[C2_4]], <16 x float> [[S1_4]], <16 x float> [[B]]
+; CHECK-NEXT: [[C3_4:%.*]] = fcmp ogt <16 x float> [[S2_4]], [[A]]
+; CHECK-NEXT: [[S3_4:%.*]] = select <16 x i1> [[C3_4]], <16 x float> [[S2_4]], <16 x float> [[B]]
+; CHECK-NEXT: [[C4_4:%.*]] = fcmp ogt <16 x float> [[S3_4]], [[A]]
+; CHECK-NEXT: [[S4_4:%.*]] = select <16 x i1> [[C4_4]], <16 x float> [[S3_4]], <16 x float> [[B]]
+; CHECK-NEXT: [[C5_4:%.*]] = fcmp ogt <16 x float> [[S4_4]], [[A]]
+; CHECK-NEXT: [[S5_4:%.*]] = select <16 x i1> [[C5_4]], <16 x float> [[S4_4]], <16 x float> [[B]]
+; CHECK-NEXT: store <16 x float> [[S5_4]], ptr addrspace(1) [[GEP_4]], align 64
+; CHECK-NEXT: [[INC_4:%.*]] = add nuw nsw i32 [[I]], 5
+; CHECK-NEXT: [[GEP_5:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC_4]]
+; CHECK-NEXT: [[V_5:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_5]], align 64
+; CHECK-NEXT: [[C1_5:%.*]] = fcmp ogt <16 x float> [[V_5]], [[A]]
+; CHECK-NEXT: [[S1_5:%.*]] = select <16 x i1> [[C1_5]], <16 x float> [[V_5]], <16 x float> [[B]]
+; CHECK-NEXT: [[C2_5:%.*]] = fcmp ogt <16 x float> [[S1_5]], [[A]]
+; CHECK-NEXT: [[S2_5:%.*]] = select <16 x i1> [[C2_5]], <16 x float> [[S1_5]], <16 x float> [[B]]
+; CHECK-NEXT: [[C3_5:%.*]] = fcmp ogt <16 x float> [[S2_5]], [[A]]
+; CHECK-NEXT: [[S3_5:%.*]] = select <16 x i1> [[C3_5]], <16 x float> [[S2_5]], <16 x float> [[B]]
+; CHECK-NEXT: [[C4_5:%.*]] = fcmp ogt <16 x float> [[S3_5]], [[A]]
+; CHECK-NEXT: [[S4_5:%.*]] = select <16 x i1> [[C4_5]], <16 x float> [[S3_5]], <16 x float> [[B]]
+; CHECK-NEXT: [[C5_5:%.*]] = fcmp ogt <16 x float> [[S4_5]], [[A]]
+; CHECK-NEXT: [[S5_5:%.*]] = select <16 x i1> [[C5_5]], <16 x float> [[S4_5]], <16 x float> [[B]]
+; CHECK-NEXT: store <16 x float> [[S5_5]], ptr addrspace(1) [[GEP_5]], align 64
+; CHECK-NEXT: [[INC_5:%.*]] = add nuw nsw i32 [[I]], 6
+; CHECK-NEXT: [[GEP_6:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC_5]]
+; CHECK-NEXT: [[V_6:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_6]], align 64
+; CHECK-NEXT: [[C1_6:%.*]] = fcmp ogt <16 x float> [[V_6]], [[A]]
+; CHECK-NEXT: [[S1_6:%.*]] = select <16 x i1> [[C1_6]], <16 x float> [[V_6]], <16 x float> [[B]]
+; CHECK-NEXT: [[C2_6:%.*]] = fcmp ogt <16 x float> [[S1_6]], [[A]]
+; CHECK-NEXT: [[S2_6:%.*]] = select <16 x i1> [[C2_6]], <16 x float> [[S1_6]], <16 x float> [[B]]
+; CHECK-NEXT: [[C3_6:%.*]] = fcmp ogt <16 x float> [[S2_6]], [[A]]
+; CHECK-NEXT: [[S3_6:%.*]] = select <16 x i1> [[C3_6]], <16 x float> [[S2_6]], <16 x float> [[B]]
+; CHECK-NEXT: [[C4_6:%.*]] = fcmp ogt <16 x float> [[S3_6]], [[A]]
+; CHECK-NEXT: [[S4_6:%.*]] = select <16 x i1> [[C4_6]], <16 x float> [[S3_6]], <16 x float> [[B]]
+; CHECK-NEXT: [[C5_6:%.*]] = fcmp ogt <16 x float> [[S4_6]], [[A]]
+; CHECK-NEXT: [[S5_6:%.*]] = select <16 x i1> [[C5_6]], <16 x float> [[S4_6]], <16 x float> [[B]]
+; CHECK-NEXT: store <16 x float> [[S5_6]], ptr addrspace(1) [[GEP_6]], align 64
+; CHECK-NEXT: [[INC_6:%.*]] = add nuw nsw i32 [[I]], 7
+; CHECK-NEXT: [[GEP_7:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC_6]]
+; CHECK-NEXT: [[V_7:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_7]], align 64
+; CHECK-NEXT: [[C1_7:%.*]] = fcmp ogt <16 x float> [[V_7]], [[A]]
+; CHECK-NEXT: [[S1_7:%.*]] = select <16 x i1> [[C1_7]], <16 x float> [[V_7]], <16 x float> [[B]]
+; CHECK-NEXT: [[C2_7:%.*]] = fcmp ogt <16 x float> [[S1_7]], [[A]]
+; CHECK-NEXT: [[S2_7:%.*]] = select <16 x i1> [[C2_7]], <16 x float> [[S1_7]], <16 x float> [[B]]
+; CHECK-NEXT: [[C3_7:%.*]] = fcmp ogt <16 x float> [[S2_7]], [[A]]
+; CHECK-NEXT: [[S3_7:%.*]] = select <16 x i1> [[C3_7]], <16 x float> [[S2_7]], <16 x float> [[B]]
+; CHECK-NEXT: [[C4_7:%.*]] = fcmp ogt <16 x float> [[S3_7]], [[A]]
+; CHECK-NEXT: [[S4_7:%.*]] = select <16 x i1> [[C4_7]], <16 x float> [[S3_7]], <16 x float> [[B]]
+; CHECK-NEXT: [[C5_7:%.*]] = fcmp ogt <16 x float> [[S4_7]], [[A]]
+; CHECK-NEXT: [[S5_7:%.*]] = select <16 x i1> [[C5_7]], <16 x float> [[S4_7]], <16 x float> [[B]]
+; CHECK-NEXT: store <16 x float> [[S5_7]], ptr addrspace(1) [[GEP_7]], align 64
+; CHECK-NEXT: [[INC_7]] = add nuw nsw i32 [[I]], 8
+; CHECK-NEXT: [[CMP_7:%.*]] = icmp samesign ult i32 [[INC_7]], 64
+; CHECK-NEXT: br i1 [[CMP_7]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+; CHECK-COUNT-5: select <16 x i1>
+entry:
+ br label %loop
+
+loop:
+ %i = phi i32 [ 0, %entry ], [ %inc, %loop ]
+ %gep = getelementptr inbounds <16 x float>, ptr addrspace(1) %p, i32 %i
+ %v = load <16 x float>, ptr addrspace(1) %gep, align 64
+ %c1 = fcmp ogt <16 x float> %v, %a
+ %s1 = select <16 x i1> %c1, <16 x float> %v, <16 x float> %b
+ %c2 = fcmp ogt <16 x float> %s1, %a
+ %s2 = select <16 x i1> %c2, <16 x float> %s1, <16 x float> %b
+ %c3 = fcmp ogt <16 x float> %s2, %a
+ %s3 = select <16 x i1> %c3, <16 x float> %s2, <16 x float> %b
+ %c4 = fcmp ogt <16 x float> %s3, %a
+ %s4 = select <16 x i1> %c4, <16 x float> %s3, <16 x float> %b
+ %c5 = fcmp ogt <16 x float> %s4, %a
+ %s5 = select <16 x i1> %c5, <16 x float> %s4, <16 x float> %b
+ store <16 x float> %s5, ptr addrspace(1) %gep, align 64
+ %inc = add nuw nsw i32 %i, 1
+ %cmp = icmp slt i32 %inc, 64
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
>From 7cafb4c72774cadd8172e404098db10059bcf2bb Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Wed, 9 Sep 2026 09:49:35 -0400
Subject: [PATCH 14/20] Use TCK_SizeAndLatency to prevent effect on unrolling
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 2 +-
.../CostModel/AMDGPU/compare-select.ll | 8 +-
.../Analysis/CostModel/AMDGPU/reduce-and.ll | 4 +-
.../Analysis/CostModel/AMDGPU/reduce-or.ll | 4 +-
.../AMDGPU/unroll-cost-cmp-select.ll | 104 +-----------------
5 files changed, 12 insertions(+), 110 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 723296380fd25..387e4f543eba4 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -998,7 +998,7 @@ InstructionCost GCNTTIImpl::getCmpSelInstrCost(
TTI::OperandValueInfo Op2Info, const Instruction *I) const {
// For size and latency cost kinds, return a low cost independent of vector
// width to enable SimplifyCFG's speculativelyExecuteBB optimization.
- if (CostKind != TTI::TCK_RecipThroughput)
+ if (CostKind == TTI::TCK_SizeAndLatency)
return 1;
return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll b/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
index 8b01d530436da..5c35879c8ec9a 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
@@ -289,7 +289,7 @@ define void @select_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
; ALL-SIZE-LABEL: 'select_v4i32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'select_v4i32'
@@ -408,7 +408,7 @@ define void @select_v3i32(i1 %cond, <3 x i32> %a, <3 x i32> %b) {
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
; ALL-SIZE-LABEL: 'select_v3i32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'select_v3i32'
@@ -476,7 +476,7 @@ define void @select_v2ptr(i1 %cond, <2 x ptr> %a, <2 x ptr> %b) {
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
; ALL-SIZE-LABEL: 'select_v2ptr'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'select_v2ptr'
@@ -493,7 +493,7 @@ define void @select_v4ptr(i1 %cond, <4 x ptr> %a, <4 x ptr> %b) {
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
; ALL-SIZE-LABEL: 'select_v4ptr'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'select_v4ptr'
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll b/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll
index b0b0b33c947b4..c9084e4c137eb 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll
@@ -24,8 +24,8 @@ define i32 @reduce_i1(i32 %arg) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V16 = call i1 @llvm.vector.reduce.and.v16i1(<16 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V32 = call i1 @llvm.vector.reduce.and.v32i1(<32 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 257 for instruction: %V64 = call i1 @llvm.vector.reduce.and.v64i1(<64 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 513 for instruction: %V128 = call i1 @llvm.vector.reduce.and.v128i1(<128 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1025 for instruction: %V256 = call i1 @llvm.vector.reduce.and.v256i1(<256 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 514 for instruction: %V128 = call i1 @llvm.vector.reduce.and.v128i1(<128 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1028 for instruction: %V256 = call i1 @llvm.vector.reduce.and.v256i1(<256 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
;
%V1 = call i1 @llvm.vector.reduce.and.v1i1(<1 x i1> undef)
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll b/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll
index 831499b4aece7..e9a6b93bea055 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll
@@ -24,8 +24,8 @@ define i32 @reduce_i1(i32 %arg) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V16 = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V32 = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 257 for instruction: %V64 = call i1 @llvm.vector.reduce.or.v64i1(<64 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 513 for instruction: %V128 = call i1 @llvm.vector.reduce.or.v128i1(<128 x i1> undef)
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1025 for instruction: %V256 = call i1 @llvm.vector.reduce.or.v256i1(<256 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 514 for instruction: %V128 = call i1 @llvm.vector.reduce.or.v128i1(<128 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1028 for instruction: %V256 = call i1 @llvm.vector.reduce.or.v256i1(<256 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret i32 undef
;
%V1 = call i1 @llvm.vector.reduce.or.v1i1(<1 x i1> undef)
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll
index 87b47ad389f07..12b8d34c97731 100644
--- a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll
@@ -11,7 +11,7 @@ define void @wide_select_loop(ptr addrspace(1) %p, <16 x float> %a, <16 x float>
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[I:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[INC_7:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[I:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[INC:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[I]]
; CHECK-NEXT: [[V:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP]], align 64
; CHECK-NEXT: [[C1:%.*]] = fcmp ogt <16 x float> [[V]], [[A]]
@@ -25,106 +25,8 @@ define void @wide_select_loop(ptr addrspace(1) %p, <16 x float> %a, <16 x float>
; CHECK-NEXT: [[C5:%.*]] = fcmp ogt <16 x float> [[S4]], [[A]]
; CHECK-NEXT: [[S5:%.*]] = select <16 x i1> [[C5]], <16 x float> [[S4]], <16 x float> [[B]]
; CHECK-NEXT: store <16 x float> [[S5]], ptr addrspace(1) [[GEP]], align 64
-; CHECK-NEXT: [[INC:%.*]] = add nuw nsw i32 [[I]], 1
-; CHECK-NEXT: [[GEP_1:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC]]
-; CHECK-NEXT: [[V_1:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_1]], align 64
-; CHECK-NEXT: [[C1_1:%.*]] = fcmp ogt <16 x float> [[V_1]], [[A]]
-; CHECK-NEXT: [[S1_1:%.*]] = select <16 x i1> [[C1_1]], <16 x float> [[V_1]], <16 x float> [[B]]
-; CHECK-NEXT: [[C2_1:%.*]] = fcmp ogt <16 x float> [[S1_1]], [[A]]
-; CHECK-NEXT: [[S2_1:%.*]] = select <16 x i1> [[C2_1]], <16 x float> [[S1_1]], <16 x float> [[B]]
-; CHECK-NEXT: [[C3_1:%.*]] = fcmp ogt <16 x float> [[S2_1]], [[A]]
-; CHECK-NEXT: [[S3_1:%.*]] = select <16 x i1> [[C3_1]], <16 x float> [[S2_1]], <16 x float> [[B]]
-; CHECK-NEXT: [[C4_1:%.*]] = fcmp ogt <16 x float> [[S3_1]], [[A]]
-; CHECK-NEXT: [[S4_1:%.*]] = select <16 x i1> [[C4_1]], <16 x float> [[S3_1]], <16 x float> [[B]]
-; CHECK-NEXT: [[C5_1:%.*]] = fcmp ogt <16 x float> [[S4_1]], [[A]]
-; CHECK-NEXT: [[S5_1:%.*]] = select <16 x i1> [[C5_1]], <16 x float> [[S4_1]], <16 x float> [[B]]
-; CHECK-NEXT: store <16 x float> [[S5_1]], ptr addrspace(1) [[GEP_1]], align 64
-; CHECK-NEXT: [[INC_1:%.*]] = add nuw nsw i32 [[I]], 2
-; CHECK-NEXT: [[GEP_2:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC_1]]
-; CHECK-NEXT: [[V_2:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_2]], align 64
-; CHECK-NEXT: [[C1_2:%.*]] = fcmp ogt <16 x float> [[V_2]], [[A]]
-; CHECK-NEXT: [[S1_2:%.*]] = select <16 x i1> [[C1_2]], <16 x float> [[V_2]], <16 x float> [[B]]
-; CHECK-NEXT: [[C2_2:%.*]] = fcmp ogt <16 x float> [[S1_2]], [[A]]
-; CHECK-NEXT: [[S2_2:%.*]] = select <16 x i1> [[C2_2]], <16 x float> [[S1_2]], <16 x float> [[B]]
-; CHECK-NEXT: [[C3_2:%.*]] = fcmp ogt <16 x float> [[S2_2]], [[A]]
-; CHECK-NEXT: [[S3_2:%.*]] = select <16 x i1> [[C3_2]], <16 x float> [[S2_2]], <16 x float> [[B]]
-; CHECK-NEXT: [[C4_2:%.*]] = fcmp ogt <16 x float> [[S3_2]], [[A]]
-; CHECK-NEXT: [[S4_2:%.*]] = select <16 x i1> [[C4_2]], <16 x float> [[S3_2]], <16 x float> [[B]]
-; CHECK-NEXT: [[C5_2:%.*]] = fcmp ogt <16 x float> [[S4_2]], [[A]]
-; CHECK-NEXT: [[S5_2:%.*]] = select <16 x i1> [[C5_2]], <16 x float> [[S4_2]], <16 x float> [[B]]
-; CHECK-NEXT: store <16 x float> [[S5_2]], ptr addrspace(1) [[GEP_2]], align 64
-; CHECK-NEXT: [[INC_2:%.*]] = add nuw nsw i32 [[I]], 3
-; CHECK-NEXT: [[GEP_3:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC_2]]
-; CHECK-NEXT: [[V_3:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_3]], align 64
-; CHECK-NEXT: [[C1_3:%.*]] = fcmp ogt <16 x float> [[V_3]], [[A]]
-; CHECK-NEXT: [[S1_3:%.*]] = select <16 x i1> [[C1_3]], <16 x float> [[V_3]], <16 x float> [[B]]
-; CHECK-NEXT: [[C2_3:%.*]] = fcmp ogt <16 x float> [[S1_3]], [[A]]
-; CHECK-NEXT: [[S2_3:%.*]] = select <16 x i1> [[C2_3]], <16 x float> [[S1_3]], <16 x float> [[B]]
-; CHECK-NEXT: [[C3_3:%.*]] = fcmp ogt <16 x float> [[S2_3]], [[A]]
-; CHECK-NEXT: [[S3_3:%.*]] = select <16 x i1> [[C3_3]], <16 x float> [[S2_3]], <16 x float> [[B]]
-; CHECK-NEXT: [[C4_3:%.*]] = fcmp ogt <16 x float> [[S3_3]], [[A]]
-; CHECK-NEXT: [[S4_3:%.*]] = select <16 x i1> [[C4_3]], <16 x float> [[S3_3]], <16 x float> [[B]]
-; CHECK-NEXT: [[C5_3:%.*]] = fcmp ogt <16 x float> [[S4_3]], [[A]]
-; CHECK-NEXT: [[S5_3:%.*]] = select <16 x i1> [[C5_3]], <16 x float> [[S4_3]], <16 x float> [[B]]
-; CHECK-NEXT: store <16 x float> [[S5_3]], ptr addrspace(1) [[GEP_3]], align 64
-; CHECK-NEXT: [[INC_3:%.*]] = add nuw nsw i32 [[I]], 4
-; CHECK-NEXT: [[GEP_4:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC_3]]
-; CHECK-NEXT: [[V_4:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_4]], align 64
-; CHECK-NEXT: [[C1_4:%.*]] = fcmp ogt <16 x float> [[V_4]], [[A]]
-; CHECK-NEXT: [[S1_4:%.*]] = select <16 x i1> [[C1_4]], <16 x float> [[V_4]], <16 x float> [[B]]
-; CHECK-NEXT: [[C2_4:%.*]] = fcmp ogt <16 x float> [[S1_4]], [[A]]
-; CHECK-NEXT: [[S2_4:%.*]] = select <16 x i1> [[C2_4]], <16 x float> [[S1_4]], <16 x float> [[B]]
-; CHECK-NEXT: [[C3_4:%.*]] = fcmp ogt <16 x float> [[S2_4]], [[A]]
-; CHECK-NEXT: [[S3_4:%.*]] = select <16 x i1> [[C3_4]], <16 x float> [[S2_4]], <16 x float> [[B]]
-; CHECK-NEXT: [[C4_4:%.*]] = fcmp ogt <16 x float> [[S3_4]], [[A]]
-; CHECK-NEXT: [[S4_4:%.*]] = select <16 x i1> [[C4_4]], <16 x float> [[S3_4]], <16 x float> [[B]]
-; CHECK-NEXT: [[C5_4:%.*]] = fcmp ogt <16 x float> [[S4_4]], [[A]]
-; CHECK-NEXT: [[S5_4:%.*]] = select <16 x i1> [[C5_4]], <16 x float> [[S4_4]], <16 x float> [[B]]
-; CHECK-NEXT: store <16 x float> [[S5_4]], ptr addrspace(1) [[GEP_4]], align 64
-; CHECK-NEXT: [[INC_4:%.*]] = add nuw nsw i32 [[I]], 5
-; CHECK-NEXT: [[GEP_5:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC_4]]
-; CHECK-NEXT: [[V_5:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_5]], align 64
-; CHECK-NEXT: [[C1_5:%.*]] = fcmp ogt <16 x float> [[V_5]], [[A]]
-; CHECK-NEXT: [[S1_5:%.*]] = select <16 x i1> [[C1_5]], <16 x float> [[V_5]], <16 x float> [[B]]
-; CHECK-NEXT: [[C2_5:%.*]] = fcmp ogt <16 x float> [[S1_5]], [[A]]
-; CHECK-NEXT: [[S2_5:%.*]] = select <16 x i1> [[C2_5]], <16 x float> [[S1_5]], <16 x float> [[B]]
-; CHECK-NEXT: [[C3_5:%.*]] = fcmp ogt <16 x float> [[S2_5]], [[A]]
-; CHECK-NEXT: [[S3_5:%.*]] = select <16 x i1> [[C3_5]], <16 x float> [[S2_5]], <16 x float> [[B]]
-; CHECK-NEXT: [[C4_5:%.*]] = fcmp ogt <16 x float> [[S3_5]], [[A]]
-; CHECK-NEXT: [[S4_5:%.*]] = select <16 x i1> [[C4_5]], <16 x float> [[S3_5]], <16 x float> [[B]]
-; CHECK-NEXT: [[C5_5:%.*]] = fcmp ogt <16 x float> [[S4_5]], [[A]]
-; CHECK-NEXT: [[S5_5:%.*]] = select <16 x i1> [[C5_5]], <16 x float> [[S4_5]], <16 x float> [[B]]
-; CHECK-NEXT: store <16 x float> [[S5_5]], ptr addrspace(1) [[GEP_5]], align 64
-; CHECK-NEXT: [[INC_5:%.*]] = add nuw nsw i32 [[I]], 6
-; CHECK-NEXT: [[GEP_6:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC_5]]
-; CHECK-NEXT: [[V_6:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_6]], align 64
-; CHECK-NEXT: [[C1_6:%.*]] = fcmp ogt <16 x float> [[V_6]], [[A]]
-; CHECK-NEXT: [[S1_6:%.*]] = select <16 x i1> [[C1_6]], <16 x float> [[V_6]], <16 x float> [[B]]
-; CHECK-NEXT: [[C2_6:%.*]] = fcmp ogt <16 x float> [[S1_6]], [[A]]
-; CHECK-NEXT: [[S2_6:%.*]] = select <16 x i1> [[C2_6]], <16 x float> [[S1_6]], <16 x float> [[B]]
-; CHECK-NEXT: [[C3_6:%.*]] = fcmp ogt <16 x float> [[S2_6]], [[A]]
-; CHECK-NEXT: [[S3_6:%.*]] = select <16 x i1> [[C3_6]], <16 x float> [[S2_6]], <16 x float> [[B]]
-; CHECK-NEXT: [[C4_6:%.*]] = fcmp ogt <16 x float> [[S3_6]], [[A]]
-; CHECK-NEXT: [[S4_6:%.*]] = select <16 x i1> [[C4_6]], <16 x float> [[S3_6]], <16 x float> [[B]]
-; CHECK-NEXT: [[C5_6:%.*]] = fcmp ogt <16 x float> [[S4_6]], [[A]]
-; CHECK-NEXT: [[S5_6:%.*]] = select <16 x i1> [[C5_6]], <16 x float> [[S4_6]], <16 x float> [[B]]
-; CHECK-NEXT: store <16 x float> [[S5_6]], ptr addrspace(1) [[GEP_6]], align 64
-; CHECK-NEXT: [[INC_6:%.*]] = add nuw nsw i32 [[I]], 7
-; CHECK-NEXT: [[GEP_7:%.*]] = getelementptr inbounds <16 x float>, ptr addrspace(1) [[P]], i32 [[INC_6]]
-; CHECK-NEXT: [[V_7:%.*]] = load <16 x float>, ptr addrspace(1) [[GEP_7]], align 64
-; CHECK-NEXT: [[C1_7:%.*]] = fcmp ogt <16 x float> [[V_7]], [[A]]
-; CHECK-NEXT: [[S1_7:%.*]] = select <16 x i1> [[C1_7]], <16 x float> [[V_7]], <16 x float> [[B]]
-; CHECK-NEXT: [[C2_7:%.*]] = fcmp ogt <16 x float> [[S1_7]], [[A]]
-; CHECK-NEXT: [[S2_7:%.*]] = select <16 x i1> [[C2_7]], <16 x float> [[S1_7]], <16 x float> [[B]]
-; CHECK-NEXT: [[C3_7:%.*]] = fcmp ogt <16 x float> [[S2_7]], [[A]]
-; CHECK-NEXT: [[S3_7:%.*]] = select <16 x i1> [[C3_7]], <16 x float> [[S2_7]], <16 x float> [[B]]
-; CHECK-NEXT: [[C4_7:%.*]] = fcmp ogt <16 x float> [[S3_7]], [[A]]
-; CHECK-NEXT: [[S4_7:%.*]] = select <16 x i1> [[C4_7]], <16 x float> [[S3_7]], <16 x float> [[B]]
-; CHECK-NEXT: [[C5_7:%.*]] = fcmp ogt <16 x float> [[S4_7]], [[A]]
-; CHECK-NEXT: [[S5_7:%.*]] = select <16 x i1> [[C5_7]], <16 x float> [[S4_7]], <16 x float> [[B]]
-; CHECK-NEXT: store <16 x float> [[S5_7]], ptr addrspace(1) [[GEP_7]], align 64
-; CHECK-NEXT: [[INC_7]] = add nuw nsw i32 [[I]], 8
-; CHECK-NEXT: [[CMP_7:%.*]] = icmp samesign ult i32 [[INC_7]], 64
+; CHECK-NEXT: [[INC]] = add nuw nsw i32 [[I]], 1
+; CHECK-NEXT: [[CMP_7:%.*]] = icmp slt i32 [[INC]], 64
; CHECK-NEXT: br i1 [[CMP_7]], label %[[LOOP]], label %[[EXIT:.*]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
>From 92e5c68e3493ed077ba6159554ba38083921dd76 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Wed, 9 Sep 2026 11:47:06 -0400
Subject: [PATCH 15/20] Fix test checks
---
.../Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll | 5 ++---
1 file changed, 2 insertions(+), 3 deletions(-)
diff --git a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll
index 12b8d34c97731..08cf3c8d94cb6 100644
--- a/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll
+++ b/llvm/test/Transforms/LoopUnroll/AMDGPU/unroll-cost-cmp-select.ll
@@ -26,12 +26,11 @@ define void @wide_select_loop(ptr addrspace(1) %p, <16 x float> %a, <16 x float>
; CHECK-NEXT: [[S5:%.*]] = select <16 x i1> [[C5]], <16 x float> [[S4]], <16 x float> [[B]]
; CHECK-NEXT: store <16 x float> [[S5]], ptr addrspace(1) [[GEP]], align 64
; CHECK-NEXT: [[INC]] = add nuw nsw i32 [[I]], 1
-; CHECK-NEXT: [[CMP_7:%.*]] = icmp slt i32 [[INC]], 64
-; CHECK-NEXT: br i1 [[CMP_7]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i32 [[INC]], 64
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT:.*]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
-; CHECK-COUNT-5: select <16 x i1>
entry:
br label %loop
>From c0bcf9f5280ca53ee88243815dbfe8d72471257c Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Thu, 10 Sep 2026 04:19:57 -0400
Subject: [PATCH 16/20] Fix merge mistakes in test expectations
---
llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll | 3 +++
llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll | 3 +++
llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll | 8 ++++----
3 files changed, 10 insertions(+), 4 deletions(-)
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll b/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll
index c9084e4c137eb..93f4bd36b6a12 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/reduce-and.ll
@@ -8,6 +8,7 @@ define i32 @reduce_i1(i32 %arg) {
; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %V1 = call i1 @llvm.vector.reduce.and.v1i1(<1 x i1> undef)
; ALL-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %V2 = call i1 @llvm.vector.reduce.and.v2i1(<2 x i1> undef)
; ALL-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V4 = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 25 for instruction: %V6 = call i1 @llvm.vector.reduce.and.v6i1(<6 x i1> poison)
; ALL-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V8 = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> undef)
; ALL-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V16 = call i1 @llvm.vector.reduce.and.v16i1(<16 x i1> undef)
; ALL-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V32 = call i1 @llvm.vector.reduce.and.v32i1(<32 x i1> undef)
@@ -20,6 +21,7 @@ define i32 @reduce_i1(i32 %arg) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %V1 = call i1 @llvm.vector.reduce.and.v1i1(<1 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %V2 = call i1 @llvm.vector.reduce.and.v2i1(<2 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V4 = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 25 for instruction: %V6 = call i1 @llvm.vector.reduce.and.v6i1(<6 x i1> poison)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V8 = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V16 = call i1 @llvm.vector.reduce.and.v16i1(<16 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V32 = call i1 @llvm.vector.reduce.and.v32i1(<32 x i1> undef)
@@ -31,6 +33,7 @@ define i32 @reduce_i1(i32 %arg) {
%V1 = call i1 @llvm.vector.reduce.and.v1i1(<1 x i1> undef)
%V2 = call i1 @llvm.vector.reduce.and.v2i1(<2 x i1> undef)
%V4 = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> undef)
+ %V6 = call i1 @llvm.vector.reduce.and.v6i1(<6 x i1> poison)
%V8 = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> undef)
%V16 = call i1 @llvm.vector.reduce.and.v16i1(<16 x i1> undef)
%V32 = call i1 @llvm.vector.reduce.and.v32i1(<32 x i1> undef)
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll b/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll
index e9a6b93bea055..eac7019bd9d82 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/reduce-or.ll
@@ -8,6 +8,7 @@ define i32 @reduce_i1(i32 %arg) {
; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %V1 = call i1 @llvm.vector.reduce.or.v1i1(<1 x i1> undef)
; ALL-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %V2 = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> undef)
; ALL-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V4 = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> undef)
+; ALL-NEXT: Cost Model: Found an estimated cost of 25 for instruction: %V6 = call i1 @llvm.vector.reduce.or.v6i1(<6 x i1> poison)
; ALL-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V8 = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> undef)
; ALL-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V16 = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> undef)
; ALL-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V32 = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> undef)
@@ -20,6 +21,7 @@ define i32 @reduce_i1(i32 %arg) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %V1 = call i1 @llvm.vector.reduce.or.v1i1(<1 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %V2 = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %V4 = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> undef)
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 25 for instruction: %V6 = call i1 @llvm.vector.reduce.or.v6i1(<6 x i1> poison)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 33 for instruction: %V8 = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 65 for instruction: %V16 = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> undef)
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 129 for instruction: %V32 = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> undef)
@@ -31,6 +33,7 @@ define i32 @reduce_i1(i32 %arg) {
%V1 = call i1 @llvm.vector.reduce.or.v1i1(<1 x i1> undef)
%V2 = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> undef)
%V4 = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> undef)
+ %V6 = call i1 @llvm.vector.reduce.or.v6i1(<6 x i1> poison)
%V8 = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> undef)
%V16 = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> undef)
%V32 = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> undef)
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll
index 64ecaca17380d..f9e1c764a52ed 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/min_max.ll
@@ -362,12 +362,12 @@ define <4 x i16> @uadd_sat_v4i16(<4 x i16> %arg0, <4 x i16> %arg1) {
; GFX8-NEXT: [[ADD_0:%.*]] = call i16 @llvm.umin.i16(i16 [[ARG0_0]], i16 [[ARG1_0]])
; GFX8-NEXT: [[ADD_1:%.*]] = call i16 @llvm.umin.i16(i16 [[ARG0_1]], i16 [[ARG1_1]])
; GFX8-NEXT: [[TMP0:%.*]] = shufflevector <4 x i16> [[ARG0]], <4 x i16> poison, <2 x i32> <i32 2, i32 3>
-; GFX8-NEXT: [[TMP1:%.*]] = shufflevector <4 x i16> [[ARG1]], <4 x i16> poison, <2 x i32> <i32 2, i32 3>
-; GFX8-NEXT: [[TMP2:%.*]] = call <2 x i16> @llvm.umin.v2i16(<2 x i16> [[TMP0]], <2 x i16> [[TMP1]])
+; GFX8-NEXT: [[TMP3:%.*]] = shufflevector <4 x i16> [[ARG1]], <4 x i16> poison, <2 x i32> <i32 2, i32 3>
+; GFX8-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.umin.v2i16(<2 x i16> [[TMP0]], <2 x i16> [[TMP3]])
; GFX8-NEXT: [[INS_0:%.*]] = insertelement <4 x i16> undef, i16 [[ADD_0]], i64 0
; GFX8-NEXT: [[INS_1:%.*]] = insertelement <4 x i16> [[INS_0]], i16 [[ADD_1]], i64 1
-; GFX8-NEXT: [[TMP3:%.*]] = shufflevector <2 x i16> [[TMP2]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; GFX8-NEXT: [[INS_31:%.*]] = shufflevector <4 x i16> [[INS_1]], <4 x i16> [[TMP3]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; GFX8-NEXT: [[TMP2:%.*]] = shufflevector <2 x i16> [[TMP1]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; GFX8-NEXT: [[INS_31:%.*]] = shufflevector <4 x i16> [[INS_1]], <4 x i16> [[TMP2]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
; GFX8-NEXT: ret <4 x i16> [[INS_31]]
;
; GFX9-LABEL: @uadd_sat_v4i16(
>From 45773989c5a16f4da84794f42c4a9dd375afce18 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Tue, 22 Sep 2026 05:03:30 -0400
Subject: [PATCH 17/20] WIP
---
.../CostModel/AMDGPU/compare-select.ll | 535 +++++-------------
1 file changed, 126 insertions(+), 409 deletions(-)
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll b/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
index 5c35879c8ec9a..b0c42c40bf7da 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
@@ -5,502 +5,219 @@
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=size-latency 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE-LATENCY %s
-define void @icmp_i32(i32 %a, i32 %b) {
-; ALL-LABEL: 'icmp_i32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i32 %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_ne = icmp ne i32 %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i32 %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_sge = icmp sge i32 %a, %b
+define void @icmp_select_i32(i32 %a, i32 %b, i32 %c) {
+; ALL-LABEL: 'icmp_select_i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt i32 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cmp, i32 %a, i32 %c
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'icmp_i32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i32 %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_ne = icmp ne i32 %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i32 %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_sge = icmp sge i32 %a, %b
+; ALL-SIZE-LABEL: 'icmp_select_i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt i32 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cmp, i32 %a, i32 %c
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'icmp_i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i32 %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_ne = icmp ne i32 %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i32 %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_sge = icmp sge i32 %a, %b
+; ALL-SIZE-LATENCY-LABEL: 'icmp_select_i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt i32 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cmp, i32 %a, i32 %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- %cmp_eq = icmp eq i32 %a, %b
- %cmp_ne = icmp ne i32 %a, %b
- %cmp_slt = icmp slt i32 %a, %b
- %cmp_sge = icmp sge i32 %a, %b
+ %cmp = icmp sgt i32 %a, %b
+ %sel = select i1 %cmp, i32 %a, i32 %c
ret void
}
-define void @icmp_i64(i64 %a, i64 %b) {
-; ALL-LABEL: 'icmp_i64'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i64 %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i64 %a, %b
+define void @fcmp_select_f32(float %a, float %b, float %c) {
+; ALL-LABEL: 'fcmp_select_f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt float %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cmp, float %a, float %c
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'icmp_i64'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i64 %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i64 %a, %b
+; ALL-SIZE-LABEL: 'fcmp_select_f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt float %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cmp, float %a, float %c
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'icmp_i64'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i64 %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i64 %a, %b
+; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt float %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cmp, float %a, float %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- %cmp_eq = icmp eq i64 %a, %b
- %cmp_slt = icmp slt i64 %a, %b
+ %cmp = fcmp ogt float %a, %b
+ %sel = select i1 %cmp, float %a, float %c
ret void
}
-define void @icmp_i16(i16 %a, i16 %b) {
-; ALL-LABEL: 'icmp_i16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i16 %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i16 %a, %b
+define void @icmp_select_v2i32(<2 x i32> %a, <2 x i32> %b,
+; ALL-LABEL: 'icmp_select_v2i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp sgt <2 x i32> %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x i32> %a, <2 x i32> %c
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'icmp_i16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i16 %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i16 %a, %b
+; ALL-SIZE-LABEL: 'icmp_select_v2i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp sgt <2 x i32> %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x i32> %a, <2 x i32> %c
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'icmp_i16'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i16 %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i16 %a, %b
+; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v2i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt <2 x i32> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <2 x i1> %cmp, <2 x i32> %a, <2 x i32> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- %cmp_eq = icmp eq i16 %a, %b
- %cmp_slt = icmp slt i16 %a, %b
+ <2 x i32> %c) {
+ %cmp = icmp sgt <2 x i32> %a, %b
+ %sel = select <2 x i1> %cmp, <2 x i32> %a, <2 x i32> %c
ret void
}
-define void @fcmp_f32(float %a, float %b) {
-; ALL-LABEL: 'fcmp_f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq float %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_one = fcmp one float %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt float %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oge = fcmp oge float %a, %b
+define void @icmp_select_v3i32(<3 x i32> %a, <3 x i32> %b,
+; ALL-LABEL: 'icmp_select_v3i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %cmp = icmp sgt <3 x i32> %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select <3 x i1> %cmp, <3 x i32> %a, <3 x i32> %c
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'fcmp_f32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq float %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_one = fcmp one float %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt float %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oge = fcmp oge float %a, %b
+; ALL-SIZE-LABEL: 'icmp_select_v3i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %cmp = icmp sgt <3 x i32> %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select <3 x i1> %cmp, <3 x i32> %a, <3 x i32> %c
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'fcmp_f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq float %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_one = fcmp one float %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt float %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oge = fcmp oge float %a, %b
+; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v3i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt <3 x i32> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <3 x i1> %cmp, <3 x i32> %a, <3 x i32> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- %cmp_oeq = fcmp oeq float %a, %b
- %cmp_one = fcmp one float %a, %b
- %cmp_olt = fcmp olt float %a, %b
- %cmp_oge = fcmp oge float %a, %b
+ <3 x i32> %c) {
+ %cmp = icmp sgt <3 x i32> %a, %b
+ %sel = select <3 x i1> %cmp, <3 x i32> %a, <3 x i32> %c
ret void
}
-define void @fcmp_f64(double %a, double %b) {
-; ALL-LABEL: 'fcmp_f64'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq double %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt double %a, %b
+define void @icmp_select_v4i32(<4 x i32> %a, <4 x i32> %b,
+; ALL-LABEL: 'icmp_select_v4i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp sgt <4 x i32> %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x i32> %a, <4 x i32> %c
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'fcmp_f64'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq double %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt double %a, %b
+; ALL-SIZE-LABEL: 'icmp_select_v4i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp sgt <4 x i32> %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x i32> %a, <4 x i32> %c
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'fcmp_f64'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq double %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt double %a, %b
+; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v4i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt <4 x i32> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x i32> %a, <4 x i32> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- %cmp_oeq = fcmp oeq double %a, %b
- %cmp_olt = fcmp olt double %a, %b
+ <4 x i32> %c) {
+ %cmp = icmp sgt <4 x i32> %a, %b
+ %sel = select <4 x i1> %cmp, <4 x i32> %a, <4 x i32> %c
ret void
}
-define void @fcmp_f16(half %a, half %b) {
-; ALL-LABEL: 'fcmp_f16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq half %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt half %a, %b
+define void @fcmp_select_v2f32(<2 x float> %a, <2 x float> %b,
+; ALL-LABEL: 'fcmp_select_v2f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = fcmp ogt <2 x float> %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x float> %a, <2 x float> %c
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'fcmp_f16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq half %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt half %a, %b
+; ALL-SIZE-LABEL: 'fcmp_select_v2f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = fcmp ogt <2 x float> %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x float> %a, <2 x float> %c
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'fcmp_f16'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq half %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt half %a, %b
+; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v2f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <2 x float> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <2 x i1> %cmp, <2 x float> %a, <2 x float> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- %cmp_oeq = fcmp oeq half %a, %b
- %cmp_olt = fcmp olt half %a, %b
+ <2 x float> %c) {
+ %cmp = fcmp ogt <2 x float> %a, %b
+ %sel = select <2 x i1> %cmp, <2 x float> %a, <2 x float> %c
ret void
}
-define void @select_i32(i1 %cond, i32 %a, i32 %b) {
-; ALL-LABEL: 'select_i32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
+define void @fcmp_select_v3f32(<3 x float> %a, <3 x float> %b,
+; ALL-LABEL: 'fcmp_select_v3f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %cmp = fcmp ogt <3 x float> %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'select_i32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
+; ALL-SIZE-LABEL: 'fcmp_select_v3f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %cmp = fcmp ogt <3 x float> %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'select_i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
+; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v3f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <3 x float> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- %sel = select i1 %cond, i32 %a, i32 %b
+ <3 x float> %c) {
+ %cmp = fcmp ogt <3 x float> %a, %b
+ %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
ret void
}
-define void @select_i64(i1 %cond, i64 %a, i64 %b) {
-; ALL-LABEL: 'select_i64'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
+define void @fcmp_select_v4f32(<4 x float> %a, <4 x float> %b,
+; ALL-LABEL: 'fcmp_select_v4f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'select_i64'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
+; ALL-SIZE-LABEL: 'fcmp_select_v4f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'select_i64'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
+; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v4f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- %sel = select i1 %cond, i64 %a, i64 %b
+ <4 x float> %c) {
+ %cmp = fcmp ogt <4 x float> %a, %b
+ %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
ret void
}
-define void @select_f32(i1 %cond, float %a, float %b) {
-; ALL-LABEL: 'select_f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
+define void @icmp_select_v2ptr(<2 x ptr> %a, <2 x ptr> %b, <2 x ptr> %c) {
+; ALL-LABEL: 'icmp_select_v2ptr'
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'select_f32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
+; ALL-SIZE-LABEL: 'icmp_select_v2ptr'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'select_f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
+; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v2ptr'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- %sel = select i1 %cond, float %a, float %b
+ %cmp = icmp ugt <2 x ptr> %a, %b
+ %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
ret void
}
-define void @select_f64(i1 %cond, double %a, double %b) {
-; ALL-LABEL: 'select_f64'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
+define void @icmp_select_v4ptr(<4 x ptr> %a, <4 x ptr> %b, <4 x ptr> %c) {
+; ALL-LABEL: 'icmp_select_v4ptr'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'select_f64'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
+; ALL-SIZE-LABEL: 'icmp_select_v4ptr'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'select_f64'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
+; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v4ptr'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- %sel = select i1 %cond, double %a, double %b
- ret void
-}
-
-define void @select_i16(i1 %cond, i16 %a, i16 %b) {
-; ALL-LABEL: 'select_i16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_i16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_i16'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, i16 %a, i16 %b
- ret void
-}
-
-define void @select_f16(i1 %cond, half %a, half %b) {
-; ALL-LABEL: 'select_f16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_f16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_f16'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, half %a, half %b
- ret void
-}
-
-define void @select_ptr(i1 %cond, ptr %a, ptr %b) {
-; ALL-LABEL: 'select_ptr'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, ptr %a, ptr %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_ptr'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, ptr %a, ptr %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_ptr'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, ptr %a, ptr %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, ptr %a, ptr %b
- ret void
-}
-
-define void @select_v2i32(i1 %cond, <2 x i32> %a, <2 x i32> %b) {
-; ALL-LABEL: 'select_v2i32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v2i32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v2i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
- ret void
-}
-
-define void @select_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
-; ALL-LABEL: 'select_v4i32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v4i32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v4i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
- ret void
-}
-
-define void @select_v2i16(i1 %cond, <2 x i16> %a, <2 x i16> %b) {
-; ALL-LABEL: 'select_v2i16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v2i16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v2i16'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
- ret void
-}
-
-define void @select_v4i16(i1 %cond, <4 x i16> %a, <4 x i16> %b) {
-; ALL-LABEL: 'select_v4i16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v4i16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v4i16'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
- ret void
-}
-
-define void @select_v2f32(i1 %cond, <2 x float> %a, <2 x float> %b) {
-; ALL-LABEL: 'select_v2f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v2f32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v2f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
- ret void
-}
-
-define void @select_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) {
-; ALL-LABEL: 'select_v4f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v4f32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v4f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
- ret void
-}
-
-define void @select_v2f16(i1 %cond, <2 x half> %a, <2 x half> %b) {
-; ALL-LABEL: 'select_v2f16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v2f16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v2f16'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
- ret void
-}
-
-define void @select_v4f16(i1 %cond, <4 x half> %a, <4 x half> %b) {
-; ALL-LABEL: 'select_v4f16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v4f16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v4f16'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
- ret void
-}
-
-define void @select_v3i32(i1 %cond, <3 x i32> %a, <3 x i32> %b) {
-; ALL-LABEL: 'select_v3i32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v3i32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v3i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
- ret void
-}
-
-define void @select_v3f32(i1 %cond, <3 x float> %a, <3 x float> %b) {
-; ALL-LABEL: 'select_v3f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v3f32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v3f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
- ret void
-}
-
-define void @select_v3i16(i1 %cond, <3 x i16> %a, <3 x i16> %b) {
-; ALL-LABEL: 'select_v3i16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v3i16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v3i16'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
- ret void
-}
-
-define void @select_v3f16(i1 %cond, <3 x half> %a, <3 x half> %b) {
-; ALL-LABEL: 'select_v3f16'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v3f16'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v3f16'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
- ret void
-}
-
-define void @select_v2ptr(i1 %cond, <2 x ptr> %a, <2 x ptr> %b) {
-; ALL-LABEL: 'select_v2ptr'
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v2ptr'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v2ptr'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
- ret void
-}
-
-define void @select_v4ptr(i1 %cond, <4 x ptr> %a, <4 x ptr> %b) {
-; ALL-LABEL: 'select_v4ptr'
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'select_v4ptr'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'select_v4ptr'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
+ %cmp = icmp ugt <4 x ptr> %a, %b
+ %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
ret void
}
;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
>From 4810af4a860b93eec9033ded5651dd6f36a7dbfc Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Wed, 23 Sep 2026 02:32:25 -0400
Subject: [PATCH 18/20] Add proper combined compare and select tests reflecting
baseline expectations
The old compare-select.ll added by this PR did
only test cmp and select separately.
Add tests combining both and keep the separate
cmp and select tests since there seems to be no
CostModel test coverage for those instructions.
Move them into new test files.
---
.../{compare-select.ll => cmp-select.ll} | 92 +----
llvm/test/Analysis/CostModel/AMDGPU/cmp.ll | 152 +++++++
llvm/test/Analysis/CostModel/AMDGPU/select.ll | 383 ++++++++++++++++++
3 files changed, 557 insertions(+), 70 deletions(-)
rename llvm/test/Analysis/CostModel/AMDGPU/{compare-select.ll => cmp-select.ll} (71%)
create mode 100644 llvm/test/Analysis/CostModel/AMDGPU/cmp.ll
create mode 100644 llvm/test/Analysis/CostModel/AMDGPU/select.ll
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll b/llvm/test/Analysis/CostModel/AMDGPU/cmp-select.ll
similarity index 71%
rename from llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
rename to llvm/test/Analysis/CostModel/AMDGPU/cmp-select.ll
index b0c42c40bf7da..a00e5e0c67f24 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/compare-select.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/cmp-select.ll
@@ -5,49 +5,7 @@
; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE %s
; RUN: opt -passes="print<cost-model>" -cost-kind=size-latency 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE-LATENCY %s
-define void @icmp_select_i32(i32 %a, i32 %b, i32 %c) {
-; ALL-LABEL: 'icmp_select_i32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt i32 %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cmp, i32 %a, i32 %c
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'icmp_select_i32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt i32 %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cmp, i32 %a, i32 %c
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'icmp_select_i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt i32 %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cmp, i32 %a, i32 %c
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %cmp = icmp sgt i32 %a, %b
- %sel = select i1 %cmp, i32 %a, i32 %c
- ret void
-}
-
-define void @fcmp_select_f32(float %a, float %b, float %c) {
-; ALL-LABEL: 'fcmp_select_f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt float %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cmp, float %a, float %c
-; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
-;
-; ALL-SIZE-LABEL: 'fcmp_select_f32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt float %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cmp, float %a, float %c
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
-; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt float %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cmp, float %a, float %c
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
-;
- %cmp = fcmp ogt float %a, %b
- %sel = select i1 %cmp, float %a, float %c
- ret void
-}
-
-define void @icmp_select_v2i32(<2 x i32> %a, <2 x i32> %b,
+define void @icmp_select_v2i32(<2 x i32> %a, <2 x i32> %b, <2 x i32> %c) {
; ALL-LABEL: 'icmp_select_v2i32'
; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp sgt <2 x i32> %a, %b
; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x i32> %a, <2 x i32> %c
@@ -59,17 +17,16 @@ define void @icmp_select_v2i32(<2 x i32> %a, <2 x i32> %b,
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v2i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt <2 x i32> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <2 x i1> %cmp, <2 x i32> %a, <2 x i32> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp sgt <2 x i32> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x i32> %a, <2 x i32> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- <2 x i32> %c) {
%cmp = icmp sgt <2 x i32> %a, %b
%sel = select <2 x i1> %cmp, <2 x i32> %a, <2 x i32> %c
ret void
}
-define void @icmp_select_v3i32(<3 x i32> %a, <3 x i32> %b,
+define void @icmp_select_v3i32(<3 x i32> %a, <3 x i32> %b, <3 x i32> %c) {
; ALL-LABEL: 'icmp_select_v3i32'
; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %cmp = icmp sgt <3 x i32> %a, %b
; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select <3 x i1> %cmp, <3 x i32> %a, <3 x i32> %c
@@ -81,17 +38,16 @@ define void @icmp_select_v3i32(<3 x i32> %a, <3 x i32> %b,
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v3i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt <3 x i32> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <3 x i1> %cmp, <3 x i32> %a, <3 x i32> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %cmp = icmp sgt <3 x i32> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select <3 x i1> %cmp, <3 x i32> %a, <3 x i32> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- <3 x i32> %c) {
%cmp = icmp sgt <3 x i32> %a, %b
%sel = select <3 x i1> %cmp, <3 x i32> %a, <3 x i32> %c
ret void
}
-define void @icmp_select_v4i32(<4 x i32> %a, <4 x i32> %b,
+define void @icmp_select_v4i32(<4 x i32> %a, <4 x i32> %b, <4 x i32> %c) {
; ALL-LABEL: 'icmp_select_v4i32'
; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp sgt <4 x i32> %a, %b
; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x i32> %a, <4 x i32> %c
@@ -103,17 +59,16 @@ define void @icmp_select_v4i32(<4 x i32> %a, <4 x i32> %b,
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v4i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt <4 x i32> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x i32> %a, <4 x i32> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp sgt <4 x i32> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x i32> %a, <4 x i32> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- <4 x i32> %c) {
%cmp = icmp sgt <4 x i32> %a, %b
%sel = select <4 x i1> %cmp, <4 x i32> %a, <4 x i32> %c
ret void
}
-define void @fcmp_select_v2f32(<2 x float> %a, <2 x float> %b,
+define void @fcmp_select_v2f32(<2 x float> %a, <2 x float> %b, <2 x float> %c) {
; ALL-LABEL: 'fcmp_select_v2f32'
; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = fcmp ogt <2 x float> %a, %b
; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x float> %a, <2 x float> %c
@@ -125,17 +80,16 @@ define void @fcmp_select_v2f32(<2 x float> %a, <2 x float> %b,
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v2f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <2 x float> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <2 x i1> %cmp, <2 x float> %a, <2 x float> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = fcmp ogt <2 x float> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x float> %a, <2 x float> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- <2 x float> %c) {
%cmp = fcmp ogt <2 x float> %a, %b
%sel = select <2 x i1> %cmp, <2 x float> %a, <2 x float> %c
ret void
}
-define void @fcmp_select_v3f32(<3 x float> %a, <3 x float> %b,
+define void @fcmp_select_v3f32(<3 x float> %a, <3 x float> %b, <3 x float> %c) {
; ALL-LABEL: 'fcmp_select_v3f32'
; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %cmp = fcmp ogt <3 x float> %a, %b
; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
@@ -147,17 +101,16 @@ define void @fcmp_select_v3f32(<3 x float> %a, <3 x float> %b,
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v3f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <3 x float> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %cmp = fcmp ogt <3 x float> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- <3 x float> %c) {
%cmp = fcmp ogt <3 x float> %a, %b
%sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
ret void
}
-define void @fcmp_select_v4f32(<4 x float> %a, <4 x float> %b,
+define void @fcmp_select_v4f32(<4 x float> %a, <4 x float> %b, <4 x float> %c) {
; ALL-LABEL: 'fcmp_select_v4f32'
; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
@@ -169,11 +122,10 @@ define void @fcmp_select_v4f32(<4 x float> %a, <4 x float> %b,
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v4f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- <4 x float> %c) {
%cmp = fcmp ogt <4 x float> %a, %b
%sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
ret void
@@ -191,8 +143,8 @@ define void @icmp_select_v2ptr(<2 x ptr> %a, <2 x ptr> %b, <2 x ptr> %c) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v2ptr'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp = icmp ugt <2 x ptr> %a, %b
@@ -212,8 +164,8 @@ define void @icmp_select_v4ptr(<4 x ptr> %a, <4 x ptr> %b, <4 x ptr> %c) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v4ptr'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp = icmp ugt <4 x ptr> %a, %b
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/cmp.ll b/llvm/test/Analysis/CostModel/AMDGPU/cmp.ll
new file mode 100644
index 0000000000000..7a319f51a9942
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/AMDGPU/cmp.ll
@@ -0,0 +1,152 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu8.03-amd-amdhsa < %s | FileCheck -check-prefixes=ALL,GFX8 %s
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL,GFX9 %s
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu11.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL,GFX11 %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=size-latency 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE-LATENCY %s
+
+define void @icmp_i32(i32 %a, i32 %b) {
+; ALL-LABEL: 'icmp_i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i32 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_ne = icmp ne i32 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i32 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_sge = icmp sge i32 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'icmp_i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i32 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_ne = icmp ne i32 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i32 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_sge = icmp sge i32 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'icmp_i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i32 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_ne = icmp ne i32 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i32 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_sge = icmp sge i32 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp_eq = icmp eq i32 %a, %b
+ %cmp_ne = icmp ne i32 %a, %b
+ %cmp_slt = icmp slt i32 %a, %b
+ %cmp_sge = icmp sge i32 %a, %b
+ ret void
+}
+
+define void @icmp_i64(i64 %a, i64 %b) {
+; ALL-LABEL: 'icmp_i64'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i64 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i64 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'icmp_i64'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i64 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i64 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'icmp_i64'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i64 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i64 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp_eq = icmp eq i64 %a, %b
+ %cmp_slt = icmp slt i64 %a, %b
+ ret void
+}
+
+define void @icmp_i16(i16 %a, i16 %b) {
+; ALL-LABEL: 'icmp_i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i16 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i16 %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'icmp_i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i16 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i16 %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'icmp_i16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_eq = icmp eq i16 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_slt = icmp slt i16 %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp_eq = icmp eq i16 %a, %b
+ %cmp_slt = icmp slt i16 %a, %b
+ ret void
+}
+
+define void @fcmp_f32(float %a, float %b) {
+; ALL-LABEL: 'fcmp_f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq float %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_one = fcmp one float %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt float %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oge = fcmp oge float %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'fcmp_f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq float %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_one = fcmp one float %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt float %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oge = fcmp oge float %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'fcmp_f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq float %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_one = fcmp one float %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt float %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oge = fcmp oge float %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp_oeq = fcmp oeq float %a, %b
+ %cmp_one = fcmp one float %a, %b
+ %cmp_olt = fcmp olt float %a, %b
+ %cmp_oge = fcmp oge float %a, %b
+ ret void
+}
+
+define void @fcmp_f64(double %a, double %b) {
+; ALL-LABEL: 'fcmp_f64'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq double %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt double %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'fcmp_f64'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq double %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt double %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'fcmp_f64'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq double %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt double %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp_oeq = fcmp oeq double %a, %b
+ %cmp_olt = fcmp olt double %a, %b
+ ret void
+}
+
+define void @fcmp_f16(half %a, half %b) {
+; ALL-LABEL: 'fcmp_f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq half %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt half %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'fcmp_f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq half %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt half %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'fcmp_f16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_oeq = fcmp oeq half %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp_olt = fcmp olt half %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp_oeq = fcmp oeq half %a, %b
+ %cmp_olt = fcmp olt half %a, %b
+ ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX11: {{.*}}
+; GFX8: {{.*}}
+; GFX9: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/select.ll b/llvm/test/Analysis/CostModel/AMDGPU/select.ll
new file mode 100644
index 0000000000000..4ce37647516d5
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/AMDGPU/select.ll
@@ -0,0 +1,383 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu8.03-amd-amdhsa < %s | FileCheck -check-prefixes=ALL,GFX8 %s
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL,GFX9 %s
+; RUN: opt -passes="print<cost-model>" 2>&1 -disable-output -mtriple=amdgpu11.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL,GFX11 %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=code-size 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE %s
+; RUN: opt -passes="print<cost-model>" -cost-kind=size-latency 2>&1 -disable-output -mtriple=amdgpu9.00-amd-amdhsa < %s | FileCheck -check-prefixes=ALL-SIZE-LATENCY %s
+
+define void @select_i32(i1 %cond, i32 %a, i32 %b) {
+; ALL-LABEL: 'select_i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'fcmp_select_v3f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %cmp = fcmp ogt <3 x float> %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v3f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <3 x float> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ <3 x float> %c) {
+ %cmp = fcmp ogt <3 x float> %a, %b
+ %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
+ ret void
+}
+
+define void @fcmp_select_v4f32(<4 x float> %a, <4 x float> %b,
+; ALL-LABEL: 'fcmp_select_v4f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'fcmp_select_v4f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v4f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ <4 x float> %c) {
+ %cmp = fcmp ogt <4 x float> %a, %b
+ %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
+ ret void
+}
+
+define void @icmp_select_v2ptr(<2 x ptr> %a, <2 x ptr> %b, <2 x ptr> %c) {
+; ALL-LABEL: 'icmp_select_v2ptr'
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'icmp_select_v2ptr'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v2ptr'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %cmp = icmp ugt <2 x ptr> %a, %b
+ %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
+ ret void
+}
+
+define void @icmp_select_v4ptr(<4 x ptr> %a, <4 x ptr> %b, <4 x ptr> %c) {
+; ALL-LABEL: 'icmp_select_v4ptr'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'icmp_select_v4ptr'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v4ptr'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, double %a, double %b
+ ret void
+}
+
+define void @select_i16(i1 %cond, i16 %a, i16 %b) {
+; ALL-LABEL: 'select_i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_i16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i16 %a, i16 %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, i16 %a, i16 %b
+ ret void
+}
+
+define void @select_f16(i1 %cond, half %a, half %b) {
+; ALL-LABEL: 'select_f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_f16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, half %a, half %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, half %a, half %b
+ ret void
+}
+
+define void @select_ptr(i1 %cond, ptr %a, ptr %b) {
+; ALL-LABEL: 'select_ptr'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, ptr %a, ptr %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_ptr'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, ptr %a, ptr %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_ptr'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, ptr %a, ptr %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, ptr %a, ptr %b
+ ret void
+}
+
+define void @select_v2i32(i1 %cond, <2 x i32> %a, <2 x i32> %b) {
+; ALL-LABEL: 'select_v2i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v2i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x i32> %a, <2 x i32> %b
+ ret void
+}
+
+define void @select_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
+; ALL-LABEL: 'select_v4i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v4i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+ ret void
+}
+
+define void @select_v2i16(i1 %cond, <2 x i16> %a, <2 x i16> %b) {
+; ALL-LABEL: 'select_v2i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v2i16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x i16> %a, <2 x i16> %b
+ ret void
+}
+
+define void @select_v4i16(i1 %cond, <4 x i16> %a, <4 x i16> %b) {
+; ALL-LABEL: 'select_v4i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v4i16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x i16> %a, <4 x i16> %b
+ ret void
+}
+
+define void @select_v2f32(i1 %cond, <2 x float> %a, <2 x float> %b) {
+; ALL-LABEL: 'select_v2f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v2f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x float> %a, <2 x float> %b
+ ret void
+}
+
+define void @select_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) {
+; ALL-LABEL: 'select_v4f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v4f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x float> %a, <4 x float> %b
+ ret void
+}
+
+define void @select_v2f16(i1 %cond, <2 x half> %a, <2 x half> %b) {
+; ALL-LABEL: 'select_v2f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v2f16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x half> %a, <2 x half> %b
+ ret void
+}
+
+define void @select_v4f16(i1 %cond, <4 x half> %a, <4 x half> %b) {
+; ALL-LABEL: 'select_v4f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v4f16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x half> %a, <4 x half> %b
+ ret void
+}
+
+define void @select_v3i32(i1 %cond, <3 x i32> %a, <3 x i32> %b) {
+; ALL-LABEL: 'select_v3i32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v3i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v3i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
+ ret void
+}
+
+define void @select_v3f32(i1 %cond, <3 x float> %a, <3 x float> %b) {
+; ALL-LABEL: 'select_v3f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v3f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v3f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <3 x float> %a, <3 x float> %b
+ ret void
+}
+
+define void @select_v3i16(i1 %cond, <3 x i16> %a, <3 x i16> %b) {
+; ALL-LABEL: 'select_v3i16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v3i16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v3i16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <3 x i16> %a, <3 x i16> %b
+ ret void
+}
+
+define void @select_v3f16(i1 %cond, <3 x half> %a, <3 x half> %b) {
+; ALL-LABEL: 'select_v3f16'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v3f16'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v3f16'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <3 x half> %a, <3 x half> %b
+ ret void
+}
+
+define void @select_v2ptr(i1 %cond, <2 x ptr> %a, <2 x ptr> %b) {
+; ALL-LABEL: 'select_v2ptr'
+; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v2ptr'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v2ptr'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
+ ret void
+}
+
+define void @select_v4ptr(i1 %cond, <4 x ptr> %a, <4 x ptr> %b) {
+; ALL-LABEL: 'select_v4ptr'
+; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
+; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; ALL-SIZE-LABEL: 'select_v4ptr'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; ALL-SIZE-LATENCY-LABEL: 'select_v4ptr'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
+ ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX11: {{.*}}
+; GFX8: {{.*}}
+; GFX9: {{.*}}
>From edcdb7b75d6f1e726d52e081854fb8f6aaaa9451 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Wed, 23 Sep 2026 04:28:19 -0400
Subject: [PATCH 19/20] Update test checks
---
.../Analysis/CostModel/AMDGPU/cmp-select.ll | 32 ++++----
llvm/test/Analysis/CostModel/AMDGPU/select.ll | 80 ++++++++-----------
2 files changed, 48 insertions(+), 64 deletions(-)
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/cmp-select.ll b/llvm/test/Analysis/CostModel/AMDGPU/cmp-select.ll
index a00e5e0c67f24..84cc84f3a426e 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/cmp-select.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/cmp-select.ll
@@ -17,8 +17,8 @@ define void @icmp_select_v2i32(<2 x i32> %a, <2 x i32> %b, <2 x i32> %c) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v2i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp sgt <2 x i32> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x i32> %a, <2 x i32> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt <2 x i32> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <2 x i1> %cmp, <2 x i32> %a, <2 x i32> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp = icmp sgt <2 x i32> %a, %b
@@ -38,8 +38,8 @@ define void @icmp_select_v3i32(<3 x i32> %a, <3 x i32> %b, <3 x i32> %c) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v3i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %cmp = icmp sgt <3 x i32> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select <3 x i1> %cmp, <3 x i32> %a, <3 x i32> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt <3 x i32> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <3 x i1> %cmp, <3 x i32> %a, <3 x i32> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp = icmp sgt <3 x i32> %a, %b
@@ -59,8 +59,8 @@ define void @icmp_select_v4i32(<4 x i32> %a, <4 x i32> %b, <4 x i32> %c) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v4i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp sgt <4 x i32> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x i32> %a, <4 x i32> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp sgt <4 x i32> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x i32> %a, <4 x i32> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp = icmp sgt <4 x i32> %a, %b
@@ -80,8 +80,8 @@ define void @fcmp_select_v2f32(<2 x float> %a, <2 x float> %b, <2 x float> %c) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v2f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = fcmp ogt <2 x float> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x float> %a, <2 x float> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <2 x float> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <2 x i1> %cmp, <2 x float> %a, <2 x float> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp = fcmp ogt <2 x float> %a, %b
@@ -101,8 +101,8 @@ define void @fcmp_select_v3f32(<3 x float> %a, <3 x float> %b, <3 x float> %c) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v3f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %cmp = fcmp ogt <3 x float> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <3 x float> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp = fcmp ogt <3 x float> %a, %b
@@ -122,8 +122,8 @@ define void @fcmp_select_v4f32(<4 x float> %a, <4 x float> %b, <4 x float> %c) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v4f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp = fcmp ogt <4 x float> %a, %b
@@ -143,8 +143,8 @@ define void @icmp_select_v2ptr(<2 x ptr> %a, <2 x ptr> %b, <2 x ptr> %c) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v2ptr'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp = icmp ugt <2 x ptr> %a, %b
@@ -164,8 +164,8 @@ define void @icmp_select_v4ptr(<4 x ptr> %a, <4 x ptr> %b, <4 x ptr> %c) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v4ptr'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%cmp = icmp ugt <4 x ptr> %a, %b
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/select.ll b/llvm/test/Analysis/CostModel/AMDGPU/select.ll
index 4ce37647516d5..356efcf795888 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/select.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/select.ll
@@ -10,79 +10,63 @@ define void @select_i32(i1 %cond, i32 %a, i32 %b) {
; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'fcmp_select_v3f32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %cmp = fcmp ogt <3 x float> %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
+; ALL-SIZE-LABEL: 'select_i32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v3f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <3 x float> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
+; ALL-SIZE-LATENCY-LABEL: 'select_i32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i32 %a, i32 %b
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- <3 x float> %c) {
- %cmp = fcmp ogt <3 x float> %a, %b
- %sel = select <3 x i1> %cmp, <3 x float> %a, <3 x float> %c
+ %sel = select i1 %cond, i32 %a, i32 %b
ret void
}
-define void @fcmp_select_v4f32(<4 x float> %a, <4 x float> %b,
-; ALL-LABEL: 'fcmp_select_v4f32'
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
+define void @select_i64(i1 %cond, i64 %a, i64 %b) {
+; ALL-LABEL: 'select_i64'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'fcmp_select_v4f32'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
+; ALL-SIZE-LABEL: 'select_i64'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'fcmp_select_v4f32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = fcmp ogt <4 x float> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
+; ALL-SIZE-LATENCY-LABEL: 'select_i64'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, i64 %a, i64 %b
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- <4 x float> %c) {
- %cmp = fcmp ogt <4 x float> %a, %b
- %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %c
+ %sel = select i1 %cond, i64 %a, i64 %b
ret void
}
-define void @icmp_select_v2ptr(<2 x ptr> %a, <2 x ptr> %b, <2 x ptr> %c) {
-; ALL-LABEL: 'icmp_select_v2ptr'
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
+define void @select_f32(i1 %cond, float %a, float %b) {
+; ALL-LABEL: 'select_f32'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'icmp_select_v2ptr'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
+; ALL-SIZE-LABEL: 'select_f32'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v2ptr'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp ugt <2 x ptr> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
+; ALL-SIZE-LATENCY-LABEL: 'select_f32'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, float %a, float %b
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
- %cmp = icmp ugt <2 x ptr> %a, %b
- %sel = select <2 x i1> %cmp, <2 x ptr> %a, <2 x ptr> %c
+ %sel = select i1 %cond, float %a, float %b
ret void
}
-define void @icmp_select_v4ptr(<4 x ptr> %a, <4 x ptr> %b, <4 x ptr> %c) {
-; ALL-LABEL: 'icmp_select_v4ptr'
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
-; ALL-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
+define void @select_f64(i1 %cond, double %a, double %b) {
+; ALL-LABEL: 'select_f64'
+; ALL-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
; ALL-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; ALL-SIZE-LABEL: 'icmp_select_v4ptr'
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
-; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
+; ALL-SIZE-LABEL: 'select_f64'
+; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; ALL-SIZE-LATENCY-LABEL: 'icmp_select_v4ptr'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %cmp = icmp ugt <4 x ptr> %a, %b
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select <4 x i1> %cmp, <4 x ptr> %a, <4 x ptr> %c
+; ALL-SIZE-LATENCY-LABEL: 'select_f64'
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, double %a, double %b
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, double %a, double %b
@@ -167,7 +151,7 @@ define void @select_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'select_v4i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
@@ -286,7 +270,7 @@ define void @select_v3i32(i1 %cond, <3 x i32> %a, <3 x i32> %b) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'select_v3i32'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 3 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <3 x i32> %a, <3 x i32> %b
@@ -354,7 +338,7 @@ define void @select_v2ptr(i1 %cond, <2 x ptr> %a, <2 x ptr> %b) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'select_v2ptr'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <2 x ptr> %a, <2 x ptr> %b
@@ -371,7 +355,7 @@ define void @select_v4ptr(i1 %cond, <4 x ptr> %a, <4 x ptr> %b) {
; ALL-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
; ALL-SIZE-LATENCY-LABEL: 'select_v4ptr'
-; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
+; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
; ALL-SIZE-LATENCY-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%sel = select i1 %cond, <4 x ptr> %a, <4 x ptr> %b
>From 816ba4cfc0fc4cf8888d8bb995d652cab8c03822 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Wed, 23 Sep 2026 10:33:11 -0400
Subject: [PATCH 20/20] Trigger build
More information about the llvm-commits
mailing list