[llvm] [AMDGPU] Adjust amdgpu fmax/fmin legalization for GISEL (PR #223351)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 23 18:00:55 PDT 2026
https://github.com/Shoreshen updated https://github.com/llvm/llvm-project/pull/223351
>From 42b213d2713ebd9bcb5ef603d78f4a8de291bee2 Mon Sep 17 00:00:00 2001
From: shore <shorshen at amd.com>
Date: Mon, 14 Sep 2026 18:17:08 +0800
Subject: [PATCH 1/2] max/min _num_ insts for gisel
---
.../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp | 31 +-
.../Target/AMDGPU/AMDGPURegBankCombiner.cpp | 14 +-
.../AMDGPU/AMDGPURegBankLegalizeRules.cpp | 19 +-
.../Target/AMDGPU/AMDGPURegisterBankInfo.cpp | 23 +-
.../AMDGPU/GlobalISel/atomicrmw_fmax.ll | 84 +--
.../AMDGPU/GlobalISel/atomicrmw_fmin.ll | 84 +--
.../GlobalISel/clamp-fmed3-const-combine.ll | 8 -
.../GlobalISel/fmed3-min-max-const-combine.ll | 18 -
.../AMDGPU/GlobalISel/fmin3-fmax3-combine.ll | 48 +-
.../inst-select-extendedLLTs-err.mir | 4 +-
.../GlobalISel/inst-select-extendedLLTs.mir | 18 +-
.../test/CodeGen/AMDGPU/flat-saddr-atomics.ll | 16 +-
llvm/test/CodeGen/AMDGPU/fmaxnum.ll | 421 +++--------
llvm/test/CodeGen/AMDGPU/fminnum.ll | 399 +++--------
llvm/test/CodeGen/AMDGPU/minmax.ll | 186 ++---
.../CodeGen/AMDGPU/packed-fneg-fsub-fp16.ll | 131 +---
.../packed-fp64-uniform-vgpr-splat-operand.ll | 10 +-
.../test/CodeGen/AMDGPU/vector-reduce-fmax.ll | 654 +++++-------------
.../test/CodeGen/AMDGPU/vector-reduce-fmin.ll | 654 +++++-------------
19 files changed, 769 insertions(+), 2053 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index 6ca65e6fa01af..3ff9153b9840e 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -1023,34 +1023,42 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
getActionDefinitionsBuilder({G_FMINNUM_IEEE, G_FMAXNUM_IEEE});
if (ST.hasVOP3PInsts()) {
- MinNumMaxNumIeee.legalFor(FPTypesPK16)
+ MinNumMaxNumIeee.legalFor(!ST.hasIEEEMinimumMaximumInsts(), FPTypesPK16)
.moreElementsIf(isSmallOddVector(0), oneMoreElement(0))
.clampMaxNumElements(0, F16, 2)
.scalarize(0);
} else if (ST.has16BitInsts()) {
- MinNumMaxNumIeee.legalFor(FPTypes16).scalarize(0);
+ MinNumMaxNumIeee.legalFor(!ST.hasIEEEMinimumMaximumInsts(), FPTypes16)
+ .scalarize(0);
} else {
- MinNumMaxNumIeee.legalFor(FPTypesBase).scalarize(0);
+ MinNumMaxNumIeee.legalFor(!ST.hasIEEEMinimumMaximumInsts(), FPTypesBase)
+ .scalarize(0);
}
auto &MinNumMaxNum = getActionDefinitionsBuilder(
{G_FMINNUM, G_FMAXNUM, G_FMINIMUMNUM, G_FMAXIMUMNUM});
if (ST.hasAnyPackedFP64Ops()) {
- MinNumMaxNum.customFor(FPTypesPK16_64)
+ MinNumMaxNum.legalFor(ST.hasIEEEMinimumMaximumInsts(), FPTypesPK16_64)
+ .customFor(FPTypesPK16_64)
.moreElementsIf(isSmallOddVector(0), oneMoreElement(0))
.clampMaxNumElements(0, F16, 2)
.clampMaxNumElements(0, F64, 2)
.scalarize(0);
} else if (ST.hasVOP3PInsts()) {
- MinNumMaxNum.customFor(FPTypesPK16)
+ MinNumMaxNum.legalFor(ST.hasIEEEMinimumMaximumInsts(), FPTypesPK16)
+ .customFor(FPTypesPK16)
.moreElementsIf(isSmallOddVector(0), oneMoreElement(0))
.clampMaxNumElements(0, F16, 2)
.scalarize(0);
} else if (ST.has16BitInsts()) {
- MinNumMaxNum.customFor(FPTypes16).scalarize(0);
+ MinNumMaxNum.legalFor(ST.hasIEEEMinimumMaximumInsts(), FPTypes16)
+ .customFor(FPTypes16)
+ .scalarize(0);
} else {
- MinNumMaxNum.customFor(FPTypesBase).scalarize(0);
+ MinNumMaxNum.legalFor(ST.hasIEEEMinimumMaximumInsts(), FPTypesBase)
+ .customFor(FPTypesBase)
+ .scalarize(0);
}
if (!ST.has16BitInsts()) {
@@ -4421,7 +4429,7 @@ bool AMDGPULegalizerInfo::legalizeFFloor(MachineInstr &MI,
// We don't need to concern ourselves with the snan handling difference, so
// use the one which will directly select.
const SIMachineFunctionInfo *MFI = B.getMF().getInfo<SIMachineFunctionInfo>();
- if (MFI->getMode().IEEE)
+ if (MFI->getMode().IEEE && !ST.hasIEEEMinimumMaximumInsts())
B.buildFMinNumIEEE(Min, Fract, Const, Flags);
else
B.buildFMinNum(Min, Fract, Const, Flags);
@@ -6229,12 +6237,13 @@ bool AMDGPULegalizerInfo::legalizeRsqClampIntrinsic(MachineInstr &MI,
const bool UseIEEE = MFI->getMode().IEEE;
auto MaxFlt = B.buildFConstant(Ty, APFloat::getLargest(*FltSemantics));
- auto ClampMax = UseIEEE ? B.buildFMinNumIEEE(Ty, Rsq, MaxFlt, Flags) :
- B.buildFMinNum(Ty, Rsq, MaxFlt, Flags);
+ auto ClampMax = UseIEEE && !ST.hasIEEEMinimumMaximumInsts()
+ ? B.buildFMinNumIEEE(Ty, Rsq, MaxFlt, Flags)
+ : B.buildFMinNum(Ty, Rsq, MaxFlt, Flags);
auto MinFlt = B.buildFConstant(Ty, APFloat::getLargest(*FltSemantics, true));
- if (UseIEEE)
+ if (UseIEEE && !ST.hasIEEEMinimumMaximumInsts())
B.buildFMaxNumIEEE(Dst, ClampMax, MinFlt, Flags);
else
B.buildFMaxNum(Dst, ClampMax, MinFlt, Flags);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURegBankCombiner.cpp b/llvm/lib/Target/AMDGPU/AMDGPURegBankCombiner.cpp
index 790c887950245..79208b2b44c65 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURegBankCombiner.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURegBankCombiner.cpp
@@ -109,6 +109,8 @@ class AMDGPURegBankCombinerImpl : public Combiner {
bool getIEEE() const;
bool getDX10Clamp() const;
bool isFminnumIeee(const MachineInstr &MI) const;
+ bool isFminnum(const MachineInstr &MI) const;
+ bool isFminnumLike(const MachineInstr &MI) const;
bool isFCst(MachineInstr *MI) const;
bool isClampZeroToOne(MachineInstr *K0, MachineInstr *K1) const;
@@ -280,7 +282,7 @@ bool AMDGPURegBankCombinerImpl::matchFPMinMaxToMed3(
// nodes(max/min) have same behavior when one input is NaN and other isn't.
// Don't consider max(min(SNaN, K1), K0) since there is no isKnownNeverQNaN,
// also post-legalizer inputs to min/max are fcanonicalized (never SNaN).
- if ((getIEEE() && isFminnumIeee(MI)) || VT->isKnownNeverNaN(Dst)) {
+ if ((getIEEE() && isFminnumLike(MI)) || VT->isKnownNeverNaN(Dst)) {
// Don't fold single use constant that can't be inlined.
if ((!MRI.hasOneNonDBGUse(K0->VReg) || TII.isInlineConstant(K0->Value)) &&
(!MRI.hasOneNonDBGUse(K1->VReg) || TII.isInlineConstant(K1->Value))) {
@@ -313,7 +315,7 @@ bool AMDGPURegBankCombinerImpl::matchFPMinMaxToClamp(MachineInstr &MI,
// no NaN inputs. Most often MI is marked with nnan fast math flag.
// For IEEE=true consider NaN inputs. Only min(max(QNaN, 0.0), 1.0) evaluates
// to 0.0 requires dx10_clamp = true.
- if ((getIEEE() && getDX10Clamp() && isFminnumIeee(MI) &&
+ if ((getIEEE() && getDX10Clamp() && isFminnumLike(MI) &&
VT->isKnownNeverSNaN(Val)) ||
VT->isKnownNeverNaN(MI.getOperand(0).getReg())) {
Reg = Val;
@@ -613,6 +615,14 @@ bool AMDGPURegBankCombinerImpl::isFminnumIeee(const MachineInstr &MI) const {
return MI.getOpcode() == AMDGPU::G_FMINNUM_IEEE;
}
+bool AMDGPURegBankCombinerImpl::isFminnum(const MachineInstr &MI) const {
+ return MI.getOpcode() == AMDGPU::G_FMINNUM;
+}
+
+bool AMDGPURegBankCombinerImpl::isFminnumLike(const MachineInstr &MI) const {
+ return isFminnumIeee(MI) || isFminnum(MI);
+}
+
bool AMDGPURegBankCombinerImpl::isFCst(MachineInstr *MI) const {
return MI->getOpcode() == AMDGPU::G_FCONSTANT;
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp b/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
index 8d8cd71374211..c0c3cc7a39a17 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURegBankLegalizeRules.cpp
@@ -1688,9 +1688,7 @@ RegBankLegalizeRules::RegBankLegalizeRules(const GCNSubtarget &_ST,
.Uni(V2S16, {{UniInVgprV2S16}, {VgprV2S16, VgprV2S16}})
.Div(V2S16, {{VgprV2S16}, {VgprV2S16, VgprV2S16}});
- addRulesForGOpcs({G_FMINNUM_IEEE, G_FMAXNUM_IEEE, G_FMINNUM, G_FMAXNUM,
- G_FMINIMUMNUM, G_FMAXIMUMNUM},
- Standard)
+ addRulesForGOpcs({G_FMINNUM_IEEE, G_FMAXNUM_IEEE}, Standard)
.Div(S16, {{Vgpr16}, {Vgpr16, Vgpr16}})
.Div(S32, {{Vgpr32}, {Vgpr32, Vgpr32}})
.Uni(S64, {{UniInVgprS64}, {Vgpr64, Vgpr64}})
@@ -1702,6 +1700,21 @@ RegBankLegalizeRules::RegBankLegalizeRules(const GCNSubtarget &_ST,
.Uni(S32, {{Sgpr32}, {Sgpr32, Sgpr32}}, hasSALUFloat)
.Uni(S32, {{UniInVgprS32}, {Vgpr32, Vgpr32}}, !hasSALUFloat);
+ addRulesForGOpcs({G_FMINNUM, G_FMAXNUM, G_FMINIMUMNUM, G_FMAXIMUMNUM},
+ Standard)
+ .Div(S16, {{Vgpr16}, {Vgpr16, Vgpr16}})
+ .Div(S32, {{Vgpr32}, {Vgpr32, Vgpr32}})
+ .Uni(S64, {{UniInVgprS64}, {Vgpr64, Vgpr64}})
+ .Div(S64, {{Vgpr64}, {Vgpr64, Vgpr64}})
+ .Uni(V2S16, {{UniInVgprV2S16}, {VgprV2S16, VgprV2S16}})
+ .Div(V2S16, {{VgprV2S16}, {VgprV2S16, VgprV2S16}})
+ .Uni(S16, {{Sgpr16}, {Sgpr16, Sgpr16}}, hasSALUFloat)
+ .Uni(S16, {{UniInVgprS16}, {Vgpr16, Vgpr16}}, !hasSALUFloat)
+ .Uni(S32, {{Sgpr32}, {Sgpr32, Sgpr32}}, hasSALUFloat)
+ .Uni(S32, {{UniInVgprS32}, {Vgpr32, Vgpr32}}, !hasSALUFloat)
+ .Any({{UniV2S64}, {{UniInVgprV2S64}, {VgprV2S64, VgprV2S64}}})
+ .Any({{DivV2S64}, {{VgprV2S64}, {VgprV2S64, VgprV2S64}}});
+
addRulesForGOpcs({G_FPTRUNC})
.Any({{DivS16, S32}, {{Vgpr16}, {Vgpr32}}})
.Any({{UniS32, S64}, {{UniInVgprS32}, {Vgpr64}}})
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
index 91bd0d006cc64..ce5ec99ac4036 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURegisterBankInfo.cpp
@@ -4093,10 +4093,6 @@ AMDGPURegisterBankInfo::getInstrMapping(const MachineInstr &MI) const {
case AMDGPU::G_FFLOOR:
case AMDGPU::G_FCEIL:
case AMDGPU::G_INTRINSIC_ROUNDEVEN:
- case AMDGPU::G_FMINNUM:
- case AMDGPU::G_FMAXNUM:
- case AMDGPU::G_FMINIMUMNUM:
- case AMDGPU::G_FMAXIMUMNUM:
case AMDGPU::G_INTRINSIC_TRUNC:
case AMDGPU::G_STRICT_FADD:
case AMDGPU::G_STRICT_FSUB:
@@ -4109,6 +4105,25 @@ AMDGPURegisterBankInfo::getInstrMapping(const MachineInstr &MI) const {
return getDefaultMappingSOP(MI);
return getDefaultMappingVOP(MI);
}
+ case AMDGPU::G_FMINNUM:
+ case AMDGPU::G_FMAXNUM:
+ case AMDGPU::G_FMINIMUMNUM:
+ case AMDGPU::G_FMAXIMUMNUM: {
+ LLT Ty = MRI.getType(MI.getOperand(0).getReg());
+ unsigned Size = Ty.getSizeInBits();
+ // Outside IEEE mode the mode-dependent scalar min/max (e.g. s_max_f32) has
+ // the correct numeric behavior, so a uniform op can map to SGPR. In IEEE
+ // mode those scalar ops would require explicit input canonicalization, so
+ // only map to SGPR when a scalar _num_ min/max (e.g. s_max_num_f32) exists;
+ // otherwise force VGPR to use the self-canonicalizing v_*_num_f* form.
+ const SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
+ bool ScalarNumIsSafe =
+ !MFI->getMode().IEEE || Subtarget.hasIEEEMinimumMaximumInsts();
+ if (Subtarget.hasSALUFloatInsts() && Ty.isScalar() &&
+ (Size == 32 || Size == 16) && isSALUMapping(MI) && ScalarNumIsSafe)
+ return getDefaultMappingSOP(MI);
+ return getDefaultMappingVOP(MI);
+ }
case AMDGPU::G_FMINIMUM:
case AMDGPU::G_FMAXIMUM: {
LLT Ty = MRI.getType(MI.getOperand(0).getReg());
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmax.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmax.ll
index 5444a36a542cb..aeb812c0d39e7 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmax.ll
@@ -604,15 +604,13 @@ define double @global_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: global_load_b64 v[4:5], v[0:1], off
-; GFX12-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
; GFX12-NEXT: s_mov_b32 s0, 0
; GFX12-NEXT: .LBB6_1: ; %atomicrmw.start
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX12-NEXT: v_max_num_f64_e32 v[4:5], v[6:7], v[6:7]
-; GFX12-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[2:3]
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: v_max_num_f64_e32 v[4:5], v[6:7], v[2:3]
; GFX12-NEXT: s_wait_storecnt 0x0
; GFX12-NEXT: global_atomic_cmpswap_b64 v[4:5], v[0:1], v[4:7], off th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -760,21 +758,18 @@ define void @global_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: global_load_b64 v[4:5], v[0:1], off
-; GFX12-NEXT: v_max_num_f64_e32 v[6:7], v[2:3], v[2:3]
+; GFX12-NEXT: global_load_b64 v[6:7], v[0:1], off
; GFX12-NEXT: s_mov_b32 s0, 0
; GFX12-NEXT: .LBB7_1: ; %atomicrmw.start
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: v_max_num_f64_e32 v[2:3], v[4:5], v[4:5]
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[6:7]
+; GFX12-NEXT: v_max_num_f64_e32 v[4:5], v[6:7], v[2:3]
; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_atomic_cmpswap_b64 v[2:3], v[0:1], v[2:5], off th:TH_ATOMIC_RETURN scope:SCOPE_DEV
+; GFX12-NEXT: global_atomic_cmpswap_b64 v[4:5], v[0:1], v[4:7], off th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_inv scope:SCOPE_DEV
-; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
-; GFX12-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
+; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
+; GFX12-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1194,15 +1189,13 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: flat_load_b64 v[4:5], v[0:1]
-; GFX12-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
; GFX12-NEXT: s_mov_b32 s0, 0
; GFX12-NEXT: .LBB10_1: ; %atomicrmw.start
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX12-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX12-NEXT: v_max_num_f64_e32 v[4:5], v[6:7], v[6:7]
-; GFX12-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[2:3]
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: v_max_num_f64_e32 v[4:5], v[6:7], v[2:3]
; GFX12-NEXT: s_wait_storecnt 0x0
; GFX12-NEXT: flat_atomic_cmpswap_b64 v[4:5], v[0:1], v[4:7] th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1348,21 +1341,18 @@ define void @flat_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: flat_load_b64 v[4:5], v[0:1]
-; GFX12-NEXT: v_max_num_f64_e32 v[6:7], v[2:3], v[2:3]
+; GFX12-NEXT: flat_load_b64 v[6:7], v[0:1]
; GFX12-NEXT: s_mov_b32 s0, 0
; GFX12-NEXT: .LBB11_1: ; %atomicrmw.start
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-NEXT: v_max_num_f64_e32 v[2:3], v[4:5], v[4:5]
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[6:7]
+; GFX12-NEXT: v_max_num_f64_e32 v[4:5], v[6:7], v[2:3]
; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: flat_atomic_cmpswap_b64 v[2:3], v[0:1], v[2:5] th:TH_ATOMIC_RETURN scope:SCOPE_DEV
+; GFX12-NEXT: flat_atomic_cmpswap_b64 v[4:5], v[0:1], v[4:7] th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX12-NEXT: global_inv scope:SCOPE_DEV
-; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
-; GFX12-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
+; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
+; GFX12-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1822,32 +1812,29 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_mov_b32_e32 v8, s16
-; GFX12-NEXT: v_max_num_f64_e32 v[6:7], v[0:1], v[0:1]
-; GFX12-NEXT: buffer_load_b64 v[2:3], v8, s[0:3], null offen
+; GFX12-NEXT: v_mov_b32_e32 v10, s16
+; GFX12-NEXT: v_dual_mov_b32 v4, v0 :: v_dual_mov_b32 v5, v1
+; GFX12-NEXT: buffer_load_b64 v[0:1], v10, s[0:3], null offen
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: v_readfirstlane_b32 s4, v2
-; GFX12-NEXT: v_readfirstlane_b32 s5, v3
+; GFX12-NEXT: v_readfirstlane_b32 s4, v0
+; GFX12-NEXT: v_readfirstlane_b32 s5, v1
; GFX12-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT: v_dual_mov_b32 v4, s4 :: v_dual_mov_b32 v5, s5
+; GFX12-NEXT: v_dual_mov_b32 v8, s4 :: v_dual_mov_b32 v9, s5
; GFX12-NEXT: s_mov_b32 s4, 0
; GFX12-NEXT: .LBB14_1: ; %atomicrmw.start
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
-; GFX12-NEXT: v_max_num_f64_e32 v[0:1], v[4:5], v[4:5]
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX12-NEXT: v_max_num_f64_e32 v[6:7], v[8:9], v[4:5]
+; GFX12-NEXT: v_dual_mov_b32 v2, v8 :: v_dual_mov_b32 v3, v9
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: v_max_num_f64_e32 v[2:3], v[0:1], v[6:7]
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
-; GFX12-NEXT: v_mov_b32_e32 v2, v4
-; GFX12-NEXT: v_mov_b32_e32 v3, v5
-; GFX12-NEXT: buffer_atomic_cmpswap_b64 v[0:3], v8, s[0:3], null offen th:TH_ATOMIC_RETURN
+; GFX12-NEXT: v_dual_mov_b32 v0, v6 :: v_dual_mov_b32 v1, v7
+; GFX12-NEXT: buffer_atomic_cmpswap_b64 v[0:3], v10, s[0:3], null offen th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_inv scope:SCOPE_DEV
-; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[4:5]
-; GFX12-NEXT: v_dual_mov_b32 v5, v1 :: v_dual_mov_b32 v4, v0
+; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
+; GFX12-NEXT: v_dual_mov_b32 v9, v1 :: v_dual_mov_b32 v8, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
@@ -2010,30 +1997,27 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: v_mov_b32_e32 v6, s16
-; GFX12-NEXT: v_max_num_f64_e32 v[4:5], v[0:1], v[0:1]
; GFX12-NEXT: buffer_load_b64 v[2:3], v6, s[0:3], null offen
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_readfirstlane_b32 s4, v2
; GFX12-NEXT: v_readfirstlane_b32 s5, v3
; GFX12-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT: v_dual_mov_b32 v2, s4 :: v_dual_mov_b32 v3, s5
+; GFX12-NEXT: v_dual_mov_b32 v4, s4 :: v_dual_mov_b32 v5, s5
; GFX12-NEXT: s_mov_b32 s4, 0
; GFX12-NEXT: .LBB15_1: ; %atomicrmw.start
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
-; GFX12-NEXT: v_max_num_f64_e32 v[0:1], v[2:3], v[2:3]
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX12-NEXT: v_max_num_f64_e32 v[2:3], v[4:5], v[0:1]
+; GFX12-NEXT: v_dual_mov_b32 v10, v5 :: v_dual_mov_b32 v9, v4
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[4:5]
-; GFX12-NEXT: v_dual_mov_b32 v10, v3 :: v_dual_mov_b32 v9, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
-; GFX12-NEXT: v_dual_mov_b32 v8, v1 :: v_dual_mov_b32 v7, v0
+; GFX12-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v7, v2
; GFX12-NEXT: buffer_atomic_cmpswap_b64 v[7:10], v6, s[0:3], null offen th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_inv scope:SCOPE_DEV
-; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[7:8], v[2:3]
-; GFX12-NEXT: v_dual_mov_b32 v2, v7 :: v_dual_mov_b32 v3, v8
+; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[7:8], v[4:5]
+; GFX12-NEXT: v_dual_mov_b32 v4, v7 :: v_dual_mov_b32 v5, v8
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmin.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmin.ll
index 508e8e1da7b5e..6c03df9df23b9 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmin.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmin.ll
@@ -604,15 +604,13 @@ define double @global_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: global_load_b64 v[4:5], v[0:1], off
-; GFX12-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
; GFX12-NEXT: s_mov_b32 s0, 0
; GFX12-NEXT: .LBB6_1: ; %atomicrmw.start
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX12-NEXT: v_max_num_f64_e32 v[4:5], v[6:7], v[6:7]
-; GFX12-NEXT: v_min_num_f64_e32 v[4:5], v[4:5], v[2:3]
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: v_min_num_f64_e32 v[4:5], v[6:7], v[2:3]
; GFX12-NEXT: s_wait_storecnt 0x0
; GFX12-NEXT: global_atomic_cmpswap_b64 v[4:5], v[0:1], v[4:7], off th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -760,21 +758,18 @@ define void @global_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: global_load_b64 v[4:5], v[0:1], off
-; GFX12-NEXT: v_max_num_f64_e32 v[6:7], v[2:3], v[2:3]
+; GFX12-NEXT: global_load_b64 v[6:7], v[0:1], off
; GFX12-NEXT: s_mov_b32 s0, 0
; GFX12-NEXT: .LBB7_1: ; %atomicrmw.start
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: v_max_num_f64_e32 v[2:3], v[4:5], v[4:5]
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT: v_min_num_f64_e32 v[2:3], v[2:3], v[6:7]
+; GFX12-NEXT: v_min_num_f64_e32 v[4:5], v[6:7], v[2:3]
; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_atomic_cmpswap_b64 v[2:3], v[0:1], v[2:5], off th:TH_ATOMIC_RETURN scope:SCOPE_DEV
+; GFX12-NEXT: global_atomic_cmpswap_b64 v[4:5], v[0:1], v[4:7], off th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_inv scope:SCOPE_DEV
-; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
-; GFX12-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
+; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
+; GFX12-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1194,15 +1189,13 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: flat_load_b64 v[4:5], v[0:1]
-; GFX12-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
; GFX12-NEXT: s_mov_b32 s0, 0
; GFX12-NEXT: .LBB10_1: ; %atomicrmw.start
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX12-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX12-NEXT: v_max_num_f64_e32 v[4:5], v[6:7], v[6:7]
-; GFX12-NEXT: v_min_num_f64_e32 v[4:5], v[4:5], v[2:3]
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: v_min_num_f64_e32 v[4:5], v[6:7], v[2:3]
; GFX12-NEXT: s_wait_storecnt 0x0
; GFX12-NEXT: flat_atomic_cmpswap_b64 v[4:5], v[0:1], v[4:7] th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1348,21 +1341,18 @@ define void @flat_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: flat_load_b64 v[4:5], v[0:1]
-; GFX12-NEXT: v_max_num_f64_e32 v[6:7], v[2:3], v[2:3]
+; GFX12-NEXT: flat_load_b64 v[6:7], v[0:1]
; GFX12-NEXT: s_mov_b32 s0, 0
; GFX12-NEXT: .LBB11_1: ; %atomicrmw.start
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-NEXT: v_max_num_f64_e32 v[2:3], v[4:5], v[4:5]
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT: v_min_num_f64_e32 v[2:3], v[2:3], v[6:7]
+; GFX12-NEXT: v_min_num_f64_e32 v[4:5], v[6:7], v[2:3]
; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: flat_atomic_cmpswap_b64 v[2:3], v[0:1], v[2:5] th:TH_ATOMIC_RETURN scope:SCOPE_DEV
+; GFX12-NEXT: flat_atomic_cmpswap_b64 v[4:5], v[0:1], v[4:7] th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX12-NEXT: global_inv scope:SCOPE_DEV
-; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
-; GFX12-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
+; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
+; GFX12-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1822,32 +1812,29 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_mov_b32_e32 v8, s16
-; GFX12-NEXT: v_max_num_f64_e32 v[6:7], v[0:1], v[0:1]
-; GFX12-NEXT: buffer_load_b64 v[2:3], v8, s[0:3], null offen
+; GFX12-NEXT: v_mov_b32_e32 v10, s16
+; GFX12-NEXT: v_dual_mov_b32 v4, v0 :: v_dual_mov_b32 v5, v1
+; GFX12-NEXT: buffer_load_b64 v[0:1], v10, s[0:3], null offen
; GFX12-NEXT: s_wait_loadcnt 0x0
-; GFX12-NEXT: v_readfirstlane_b32 s4, v2
-; GFX12-NEXT: v_readfirstlane_b32 s5, v3
+; GFX12-NEXT: v_readfirstlane_b32 s4, v0
+; GFX12-NEXT: v_readfirstlane_b32 s5, v1
; GFX12-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT: v_dual_mov_b32 v4, s4 :: v_dual_mov_b32 v5, s5
+; GFX12-NEXT: v_dual_mov_b32 v8, s4 :: v_dual_mov_b32 v9, s5
; GFX12-NEXT: s_mov_b32 s4, 0
; GFX12-NEXT: .LBB14_1: ; %atomicrmw.start
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
-; GFX12-NEXT: v_max_num_f64_e32 v[0:1], v[4:5], v[4:5]
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX12-NEXT: v_min_num_f64_e32 v[6:7], v[8:9], v[4:5]
+; GFX12-NEXT: v_dual_mov_b32 v2, v8 :: v_dual_mov_b32 v3, v9
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: v_min_num_f64_e32 v[2:3], v[0:1], v[6:7]
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
-; GFX12-NEXT: v_mov_b32_e32 v2, v4
-; GFX12-NEXT: v_mov_b32_e32 v3, v5
-; GFX12-NEXT: buffer_atomic_cmpswap_b64 v[0:3], v8, s[0:3], null offen th:TH_ATOMIC_RETURN
+; GFX12-NEXT: v_dual_mov_b32 v0, v6 :: v_dual_mov_b32 v1, v7
+; GFX12-NEXT: buffer_atomic_cmpswap_b64 v[0:3], v10, s[0:3], null offen th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_inv scope:SCOPE_DEV
-; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[4:5]
-; GFX12-NEXT: v_dual_mov_b32 v5, v1 :: v_dual_mov_b32 v4, v0
+; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
+; GFX12-NEXT: v_dual_mov_b32 v9, v1 :: v_dual_mov_b32 v8, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
@@ -2010,30 +1997,27 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: v_mov_b32_e32 v6, s16
-; GFX12-NEXT: v_max_num_f64_e32 v[4:5], v[0:1], v[0:1]
; GFX12-NEXT: buffer_load_b64 v[2:3], v6, s[0:3], null offen
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_readfirstlane_b32 s4, v2
; GFX12-NEXT: v_readfirstlane_b32 s5, v3
; GFX12-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT: v_dual_mov_b32 v2, s4 :: v_dual_mov_b32 v3, s5
+; GFX12-NEXT: v_dual_mov_b32 v4, s4 :: v_dual_mov_b32 v5, s5
; GFX12-NEXT: s_mov_b32 s4, 0
; GFX12-NEXT: .LBB15_1: ; %atomicrmw.start
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
-; GFX12-NEXT: v_max_num_f64_e32 v[0:1], v[2:3], v[2:3]
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX12-NEXT: v_min_num_f64_e32 v[2:3], v[4:5], v[0:1]
+; GFX12-NEXT: v_dual_mov_b32 v10, v5 :: v_dual_mov_b32 v9, v4
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[4:5]
-; GFX12-NEXT: v_dual_mov_b32 v10, v3 :: v_dual_mov_b32 v9, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
-; GFX12-NEXT: v_dual_mov_b32 v8, v1 :: v_dual_mov_b32 v7, v0
+; GFX12-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v7, v2
; GFX12-NEXT: buffer_atomic_cmpswap_b64 v[7:10], v6, s[0:3], null offen th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_inv scope:SCOPE_DEV
-; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[7:8], v[2:3]
-; GFX12-NEXT: v_dual_mov_b32 v2, v7 :: v_dual_mov_b32 v3, v8
+; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[7:8], v[4:5]
+; GFX12-NEXT: v_dual_mov_b32 v4, v7 :: v_dual_mov_b32 v5, v8
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/clamp-fmed3-const-combine.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/clamp-fmed3-const-combine.ll
index a33785e16722a..34780a10a1c2d 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/clamp-fmed3-const-combine.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/clamp-fmed3-const-combine.ll
@@ -80,8 +80,6 @@ define float @test_fmed3_non_SNaN_input_ieee_true_dx10clamp_true(float %a) #2 {
; GFX1170-LABEL: test_fmed3_non_SNaN_input_ieee_true_dx10clamp_true:
; GFX1170: ; %bb.0:
; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-NEXT: v_max_num_f32_e32 v0, v0, v0
-; GFX1170-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-NEXT: v_min_num_f32_e64 v0, 0x41200000, v0 clamp
; GFX1170-NEXT: s_setpc_b64 s[30:31]
;
@@ -92,8 +90,6 @@ define float @test_fmed3_non_SNaN_input_ieee_true_dx10clamp_true(float %a) #2 {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_max_num_f32_e32 v0, v0, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_min_num_f32_e64 v0, 0x41200000, v0 clamp
; GFX12-NEXT: s_setpc_b64 s[30:31]
%fmin = call float @llvm.minnum.f32(float %a, float 10.0)
@@ -204,8 +200,6 @@ define float @test_fmed3_non_SNaN_input_ieee_true_dx10clamp_false(float %a) #4 {
; GFX1170-LABEL: test_fmed3_non_SNaN_input_ieee_true_dx10clamp_false:
; GFX1170: ; %bb.0:
; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-NEXT: v_max_num_f32_e32 v0, v0, v0
-; GFX1170-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-NEXT: v_min_num_f32_e64 v0, 0x41200000, v0 clamp
; GFX1170-NEXT: s_setpc_b64 s[30:31]
;
@@ -216,8 +210,6 @@ define float @test_fmed3_non_SNaN_input_ieee_true_dx10clamp_false(float %a) #4 {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_max_num_f32_e32 v0, v0, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_min_num_f32_e64 v0, 0x41200000, v0 clamp
; GFX12-NEXT: s_setpc_b64 s[30:31]
%fmin = call float @llvm.minnum.f32(float %a, float 10.0)
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fmed3-min-max-const-combine.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fmed3-min-max-const-combine.ll
index 34d4e6d60d35a..7d4191615bee8 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fmed3-min-max-const-combine.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fmed3-min-max-const-combine.ll
@@ -92,8 +92,6 @@ define half @test_min_K1max_ValK0_f16(half %a) #0 {
; GFX1170-LABEL: test_min_K1max_ValK0_f16:
; GFX1170: ; %bb.0:
; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX1170-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-NEXT: v_med3_num_f16 v0, v0, 2.0, 4.0
; GFX1170-NEXT: s_setpc_b64 s[30:31]
;
@@ -104,8 +102,6 @@ define half @test_min_K1max_ValK0_f16(half %a) #0 {
; GFX12-TRUE16-NEXT: s_wait_samplecnt 0x0
; GFX12-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_med3_num_f16 v0.l, v0.l, 2.0, 4.0
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -116,8 +112,6 @@ define half @test_min_K1max_ValK0_f16(half %a) #0 {
; GFX12-FAKE16-NEXT: s_wait_samplecnt 0x0
; GFX12-FAKE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-FAKE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_med3_num_f16 v0, v0, 2.0, 4.0
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
%maxnum = call half @llvm.maxnum.f16(half %a, half 2.0)
@@ -609,8 +603,6 @@ define float @test_min_max_maybe_NaN_input_ieee_false(float %a) #1 {
; GFX1170-LABEL: test_min_max_maybe_NaN_input_ieee_false:
; GFX1170: ; %bb.0:
; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-NEXT: v_max_num_f32_e32 v0, v0, v0
-; GFX1170-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-NEXT: v_med3_num_f32 v0, v0, 2.0, 4.0
; GFX1170-NEXT: s_setpc_b64 s[30:31]
;
@@ -621,8 +613,6 @@ define float @test_min_max_maybe_NaN_input_ieee_false(float %a) #1 {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_max_num_f32_e32 v0, v0, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_med3_num_f32 v0, v0, 2.0, 4.0
; GFX12-NEXT: s_setpc_b64 s[30:31]
%maxnum = call float @llvm.maxnum.f32(float %a, float 2.0)
@@ -648,8 +638,6 @@ define float @test_max_min_maybe_NaN_input_ieee_false(float %a) #1 {
; GFX1170-LABEL: test_max_min_maybe_NaN_input_ieee_false:
; GFX1170: ; %bb.0:
; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-NEXT: v_max_num_f32_e32 v0, v0, v0
-; GFX1170-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-NEXT: v_med3_num_f32 v0, v0, 2.0, 4.0
; GFX1170-NEXT: s_setpc_b64 s[30:31]
;
@@ -660,8 +648,6 @@ define float @test_max_min_maybe_NaN_input_ieee_false(float %a) #1 {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_max_num_f32_e32 v0, v0, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_med3_num_f32 v0, v0, 2.0, 4.0
; GFX12-NEXT: s_setpc_b64 s[30:31]
%minnum = call float @llvm.minnum.f32(float %a, float 4.0)
@@ -688,8 +674,6 @@ define float @test_max_min_maybe_NaN_input_ieee_true(float %a) #0 {
; GFX1170-LABEL: test_max_min_maybe_NaN_input_ieee_true:
; GFX1170: ; %bb.0:
; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-NEXT: v_max_num_f32_e32 v0, v0, v0
-; GFX1170-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-NEXT: v_med3_num_f32 v0, v0, 2.0, 4.0
; GFX1170-NEXT: s_setpc_b64 s[30:31]
;
@@ -700,8 +684,6 @@ define float @test_max_min_maybe_NaN_input_ieee_true(float %a) #0 {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_max_num_f32_e32 v0, v0, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_med3_num_f32 v0, v0, 2.0, 4.0
; GFX12-NEXT: s_setpc_b64 s[30:31]
%minnum = call float @llvm.minnum.f32(float %a, float 4.0)
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fmin3-fmax3-combine.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fmin3-fmax3-combine.ll
index 7b5d1b02c07df..ee43a83cb4cdd 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fmin3-fmax3-combine.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fmin3-fmax3-combine.ll
@@ -18,9 +18,6 @@ define float @test_fmin3(float %a, float %b, float %c) {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-NEXT: v_max_num_f32_e32 v2, v2, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_min3_num_f32 v0, v0, v1, v2
; GFX12-NEXT: s_setpc_b64 s[30:31]
%min1 = call float @llvm.minnum.f32(float %a, float %b)
@@ -45,19 +42,11 @@ define float @test_fmin3_inreg(float inreg %a, float inreg %b, float inreg %c) {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_max_num_f32_e64 v0, s0, s0
-; GFX12-NEXT: v_max_num_f32_e64 v1, s1, s1
-; GFX12-NEXT: v_max_num_f32_e64 v2, s2, s2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX12-NEXT: v_readfirstlane_b32 s0, v0
-; GFX12-NEXT: v_readfirstlane_b32 s1, v1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
-; GFX12-NEXT: v_readfirstlane_b32 s2, v2
; GFX12-NEXT: s_min_num_f32 s0, s0, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-NEXT: s_min_num_f32 s0, s0, s2
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-NEXT: v_mov_b32_e32 v0, s0
; GFX12-NEXT: s_setpc_b64 s[30:31]
%min1 = call float @llvm.minnum.f32(float %a, float %b)
@@ -129,8 +118,6 @@ define float @test_fmin3_with_constants(float %a, float %b) {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_min3_num_f32 v0, v0, v1, 0x40e00000
; GFX12-NEXT: s_setpc_b64 s[30:31]
%min1 = call float @llvm.minnum.f32(float %a, float %b)
@@ -154,11 +141,6 @@ define float @test_fmin3_with_constants_inreg(float inreg %a, float inreg %b) {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_max_num_f32_e64 v0, s0, s0
-; GFX12-NEXT: v_max_num_f32_e64 v1, s1, s1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-NEXT: v_readfirstlane_b32 s0, v0
-; GFX12-NEXT: v_readfirstlane_b32 s1, v1
; GFX12-NEXT: s_min_num_f32 s0, s0, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
@@ -236,9 +218,6 @@ define float @test_fmax3(float %a, float %b, float %c) {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-NEXT: v_max_num_f32_e32 v2, v2, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_max3_num_f32 v0, v0, v1, v2
; GFX12-NEXT: s_setpc_b64 s[30:31]
%max1 = call float @llvm.maxnum.f32(float %a, float %b)
@@ -263,19 +242,11 @@ define float @test_fmax3_inreg(float inreg %a, float inreg %b, float inreg %c) {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_max_num_f32_e64 v0, s0, s0
-; GFX12-NEXT: v_max_num_f32_e64 v1, s1, s1
-; GFX12-NEXT: v_max_num_f32_e64 v2, s2, s2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX12-NEXT: v_readfirstlane_b32 s0, v0
-; GFX12-NEXT: v_readfirstlane_b32 s1, v1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
-; GFX12-NEXT: v_readfirstlane_b32 s2, v2
; GFX12-NEXT: s_max_num_f32 s0, s0, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-NEXT: s_max_num_f32 s0, s0, s2
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-NEXT: v_mov_b32_e32 v0, s0
; GFX12-NEXT: s_setpc_b64 s[30:31]
%max1 = call float @llvm.maxnum.f32(float %a, float %b)
@@ -347,8 +318,6 @@ define float @test_fmax3_with_constants(float %a, float %b) {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_max3_num_f32 v0, v0, v1, 0x40e00000
; GFX12-NEXT: s_setpc_b64 s[30:31]
%max1 = call float @llvm.maxnum.f32(float %a, float %b)
@@ -372,11 +341,6 @@ define float @test_fmax3_with_constants_inreg(float inreg %a, float inreg %b) {
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_max_num_f32_e64 v0, s0, s0
-; GFX12-NEXT: v_max_num_f32_e64 v1, s1, s1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-NEXT: v_readfirstlane_b32 s0, v0
-; GFX12-NEXT: v_readfirstlane_b32 s1, v1
; GFX12-NEXT: s_max_num_f32 s0, s0, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
@@ -458,10 +422,6 @@ define <2 x float> @test_fmin3_v2f32(<2 x float> %a, <2 x float> %b, <2 x float>
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX12-NEXT: v_dual_max_num_f32 v4, v4, v4 :: v_dual_max_num_f32 v5, v5, v5
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_min3_num_f32 v0, v0, v2, v4
; GFX12-NEXT: v_min3_num_f32 v1, v1, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
@@ -491,10 +451,6 @@ define <2 x float> @test_fmax3_v2f32(<2 x float> %a, <2 x float> %b, <2 x float>
; GFX12-NEXT: s_wait_samplecnt 0x0
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX12-NEXT: v_dual_max_num_f32 v4, v4, v4 :: v_dual_max_num_f32 v5, v5, v5
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_max3_num_f32 v0, v0, v2, v4
; GFX12-NEXT: v_max3_num_f32 v1, v1, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/inst-select-extendedLLTs-err.mir b/llvm/test/CodeGen/AMDGPU/GlobalISel/inst-select-extendedLLTs-err.mir
index 79b23bec90b60..f067ab577a7d4 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/inst-select-extendedLLTs-err.mir
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/inst-select-extendedLLTs-err.mir
@@ -315,7 +315,7 @@ body: |
$vgpr0 = COPY %2
...
-# ERR: remark: <unknown>:0:0: cannot select: %2:vgpr_32(s32) = G_FMAXNUM_IEEE %0:vgpr, %1:vgpr (in function: g_fmaxnum_ieee_op0_f32)
+# ERR: remark: <unknown>:0:0: instruction is not legal: %2:vgpr(s32) = G_FMAXNUM_IEEE %0:vgpr, %1:vgpr (in function: g_fmaxnum_ieee_op0_f32)
---
name: g_fmaxnum_ieee_op0_f32
@@ -331,7 +331,7 @@ body: |
$vgpr0 = COPY %2
...
-# ERR: remark: <unknown>:0:0: cannot select: %2:vgpr_32(s32) = G_FMINNUM_IEEE %0:vgpr, %1:vgpr (in function: g_fminnum_ieee_op0_f32)
+# ERR: remark: <unknown>:0:0: instruction is not legal: %2:vgpr(s32) = G_FMINNUM_IEEE %0:vgpr, %1:vgpr (in function: g_fminnum_ieee_op0_f32)
---
name: g_fminnum_ieee_op0_f32
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/inst-select-extendedLLTs.mir b/llvm/test/CodeGen/AMDGPU/GlobalISel/inst-select-extendedLLTs.mir
index 85a69e5e2f189..68346d08c6b2a 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/inst-select-extendedLLTs.mir
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/inst-select-extendedLLTs.mir
@@ -477,6 +477,8 @@ body: |
$vgpr0 = COPY %2
...
+# G_FMAXNUM_IEEE/G_FMINNUM_IEEE not legal for gfx1251 now
+
---
name: g_fmaxnum_ieee_op0_f32
legalized: true
@@ -488,10 +490,10 @@ body: |
; GCN-LABEL: name: g_fmaxnum_ieee_op0_f32
; GCN: liveins: $vgpr0, $vgpr1
; GCN-NEXT: {{ $}}
- ; GCN-NEXT: [[COPY:%[0-9]+]]:vgpr_32 = COPY $vgpr0
- ; GCN-NEXT: [[COPY1:%[0-9]+]]:vgpr_32 = COPY $vgpr1
- ; GCN-NEXT: [[V_MAX_F32_e64_:%[0-9]+]]:vgpr_32 = nofpexcept V_MAX_F32_e64 0, [[COPY]], 0, [[COPY1]], 0, 0, implicit $mode, implicit $exec
- ; GCN-NEXT: $vgpr0 = COPY [[V_MAX_F32_e64_]]
+ ; GCN-NEXT: [[COPY:%[0-9]+]]:vgpr(s32) = COPY $vgpr0
+ ; GCN-NEXT: [[COPY1:%[0-9]+]]:vgpr(s32) = COPY $vgpr1
+ ; GCN-NEXT: [[FMAXNUM_IEEE:%[0-9]+]]:vgpr(f32) = G_FMAXNUM_IEEE [[COPY]], [[COPY1]]
+ ; GCN-NEXT: $vgpr0 = COPY [[FMAXNUM_IEEE]](f32)
%0:vgpr(s32) = COPY $vgpr0
%1:vgpr(s32) = COPY $vgpr1
%2:vgpr(f32) = G_FMAXNUM_IEEE %0, %1
@@ -509,10 +511,10 @@ body: |
; GCN-LABEL: name: g_fminnum_ieee_op0_f32
; GCN: liveins: $vgpr0, $vgpr1
; GCN-NEXT: {{ $}}
- ; GCN-NEXT: [[COPY:%[0-9]+]]:vgpr_32 = COPY $vgpr0
- ; GCN-NEXT: [[COPY1:%[0-9]+]]:vgpr_32 = COPY $vgpr1
- ; GCN-NEXT: [[V_MIN_F32_e64_:%[0-9]+]]:vgpr_32 = nofpexcept V_MIN_F32_e64 0, [[COPY]], 0, [[COPY1]], 0, 0, implicit $mode, implicit $exec
- ; GCN-NEXT: $vgpr0 = COPY [[V_MIN_F32_e64_]]
+ ; GCN-NEXT: [[COPY:%[0-9]+]]:vgpr(s32) = COPY $vgpr0
+ ; GCN-NEXT: [[COPY1:%[0-9]+]]:vgpr(s32) = COPY $vgpr1
+ ; GCN-NEXT: [[FMINNUM_IEEE:%[0-9]+]]:vgpr(f32) = G_FMINNUM_IEEE [[COPY]], [[COPY1]]
+ ; GCN-NEXT: $vgpr0 = COPY [[FMINNUM_IEEE]](f32)
%0:vgpr(s32) = COPY $vgpr0
%1:vgpr(s32) = COPY $vgpr1
%2:vgpr(f32) = G_FMINNUM_IEEE %0, %1
diff --git a/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll b/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll
index f40b209b77da7..a2c23549c4497 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll
@@ -13993,13 +13993,10 @@ define double @flat_atomic_fmax_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: s_sub_co_i32 s0, s2, src_flat_scratch_base_lo
; GFX1250-GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
-; GFX1250-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
; GFX1250-GISEL-NEXT: s_cselect_b32 s0, s0, -1
; GFX1250-GISEL-NEXT: scratch_load_b64 v[2:3], off, s0
; GFX1250-GISEL-NEXT: s_wait_loadcnt 0x0
-; GFX1250-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[2:3], v[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1250-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[4:5], v[0:1]
+; GFX1250-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[2:3], v[0:1]
; GFX1250-GISEL-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX1250-GISEL-NEXT: .LBB112_4: ; %atomicrmw.end
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
@@ -14146,12 +14143,9 @@ define void @flat_atomic_fmax_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: s_sub_co_i32 s0, s2, src_flat_scratch_base_lo
; GFX1250-GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
-; GFX1250-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
; GFX1250-GISEL-NEXT: s_cselect_b32 s0, s0, -1
; GFX1250-GISEL-NEXT: scratch_load_b64 v[2:3], off, s0
; GFX1250-GISEL-NEXT: s_wait_loadcnt 0x0
-; GFX1250-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[2:3], v[0:1]
; GFX1250-GISEL-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX1250-GISEL-NEXT: .LBB113_4: ; %atomicrmw.phi
@@ -14294,13 +14288,10 @@ define double @flat_atomic_fmin_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: s_sub_co_i32 s0, s2, src_flat_scratch_base_lo
; GFX1250-GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
-; GFX1250-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
; GFX1250-GISEL-NEXT: s_cselect_b32 s0, s0, -1
; GFX1250-GISEL-NEXT: scratch_load_b64 v[2:3], off, s0
; GFX1250-GISEL-NEXT: s_wait_loadcnt 0x0
-; GFX1250-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[2:3], v[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1250-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[4:5], v[0:1]
+; GFX1250-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[2:3], v[0:1]
; GFX1250-GISEL-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX1250-GISEL-NEXT: .LBB114_4: ; %atomicrmw.end
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
@@ -14447,12 +14438,9 @@ define void @flat_atomic_fmin_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: s_sub_co_i32 s0, s2, src_flat_scratch_base_lo
; GFX1250-GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
-; GFX1250-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
; GFX1250-GISEL-NEXT: s_cselect_b32 s0, s0, -1
; GFX1250-GISEL-NEXT: scratch_load_b64 v[2:3], off, s0
; GFX1250-GISEL-NEXT: s_wait_loadcnt 0x0
-; GFX1250-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[2:3], v[0:1]
; GFX1250-GISEL-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX1250-GISEL-NEXT: .LBB115_4: ; %atomicrmw.phi
diff --git a/llvm/test/CodeGen/AMDGPU/fmaxnum.ll b/llvm/test/CodeGen/AMDGPU/fmaxnum.ll
index 2a7198e579d42..9eb639d79b46c 100644
--- a/llvm/test/CodeGen/AMDGPU/fmaxnum.ll
+++ b/llvm/test/CodeGen/AMDGPU/fmaxnum.ll
@@ -58,35 +58,17 @@ define amdgpu_kernel void @test_fmax_f32_ieee_mode_on(ptr addrspace(1) %out, flo
; GFX9-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; GFX9-GISEL-NEXT: s_endpgm
;
-; GFX12-SDAG-LABEL: test_fmax_f32_ieee_mode_on:
-; GFX12-SDAG: ; %bb.0:
-; GFX12-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: s_max_num_f32 s2, s2, s3
-; GFX12-SDAG-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
-; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, s2
-; GFX12-SDAG-NEXT: s_mov_b32 s2, -1
-; GFX12-SDAG-NEXT: buffer_store_b32 v0, off, s[0:3], null
-; GFX12-SDAG-NEXT: s_endpgm
-;
-; GFX12-GISEL-LABEL: test_fmax_f32_ieee_mode_on:
-; GFX12-GISEL: ; %bb.0:
-; GFX12-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s2, s2
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s3, s3
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: s_max_num_f32 s2, s2, s3
-; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-NEXT: v_mov_b32_e32 v0, s2
-; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: buffer_store_b32 v0, off, s[0:3], null
-; GFX12-GISEL-NEXT: s_endpgm
+; GFX12-LABEL: test_fmax_f32_ieee_mode_on:
+; GFX12: ; %bb.0:
+; GFX12-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: s_max_num_f32 s2, s2, s3
+; GFX12-NEXT: s_mov_b32 s3, 0x31016000
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX12-NEXT: v_mov_b32_e32 v0, s2
+; GFX12-NEXT: s_mov_b32 s2, -1
+; GFX12-NEXT: buffer_store_b32 v0, off, s[0:3], null
+; GFX12-NEXT: s_endpgm
%val = call float @llvm.maxnum.f32(float %a, float %b) #1
store float %val, ptr addrspace(1) %out, align 4
ret void
@@ -199,20 +181,9 @@ define amdgpu_kernel void @test_fmax_v2f32(ptr addrspace(1) %out, <2 x float> %a
; GFX12-GISEL-NEXT: s_mov_b32 s6, -1
; GFX12-GISEL-NEXT: s_mov_b32 s7, 0x31016000
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s0, s0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s2, s2
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s1, s1
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s3, s3
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s0, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s1, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v3
-; GFX12-GISEL-NEXT: s_max_num_f32 s0, s0, s1
-; GFX12-GISEL-NEXT: s_max_num_f32 s1, s2, s3
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX12-GISEL-NEXT: s_max_num_f32 s0, s0, s2
+; GFX12-GISEL-NEXT: s_max_num_f32 s1, s1, s3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-GISEL-NEXT: buffer_store_b64 v[0:1], off, s[4:7], null
; GFX12-GISEL-NEXT: s_endpgm
@@ -320,25 +291,13 @@ define amdgpu_kernel void @test_fmax_v3f32(ptr addrspace(1) %out, <3 x float> %a
; GFX12-GISEL-NEXT: s_clause 0x1
; GFX12-GISEL-NEXT: s_load_b256 s[8:15], s[4:5], 0x34
; GFX12-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s8, s8
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s12, s12
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s9, s9
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s13, s13
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s10, s10
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v5, s14, s14
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s5, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s6, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s7, v4
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s8, v5
-; GFX12-GISEL-NEXT: s_max_num_f32 s4, s2, s3
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: s_max_num_f32 s5, s5, s6
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_max_num_f32 s6, s7, s8
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_2)
+; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT: s_max_num_f32 s4, s8, s12
+; GFX12-GISEL-NEXT: s_max_num_f32 s5, s9, s13
+; GFX12-GISEL-NEXT: s_max_num_f32 s6, s10, s14
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX12-GISEL-NEXT: v_mov_b32_e32 v2, s6
; GFX12-GISEL-NEXT: buffer_store_b96 v[0:2], off, s[0:3], null
@@ -460,32 +419,16 @@ define amdgpu_kernel void @test_fmax_v4f32(ptr addrspace(1) %out, <4 x float> %a
; GFX12-GISEL-NEXT: s_clause 0x1
; GFX12-GISEL-NEXT: s_load_b256 s[8:15], s[4:5], 0x34
; GFX12-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s8, s8
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s12, s12
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s9, s9
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s13, s13
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s10, s10
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v5, s14, s14
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v6, s11, s11
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v7, s15, s15
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s5, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s6, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s7, v4
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s8, v5
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s9, v6
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s10, v7
-; GFX12-GISEL-NEXT: s_max_num_f32 s4, s2, s3
-; GFX12-GISEL-NEXT: s_max_num_f32 s5, s5, s6
-; GFX12-GISEL-NEXT: s_max_num_f32 s6, s7, s8
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: s_max_num_f32 s7, s9, s10
+; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
+; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT: s_max_num_f32 s4, s8, s12
+; GFX12-GISEL-NEXT: s_max_num_f32 s5, s9, s13
+; GFX12-GISEL-NEXT: s_max_num_f32 s6, s10, s14
+; GFX12-GISEL-NEXT: s_max_num_f32 s7, s11, s15
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: v_dual_mov_b32 v2, s6 :: v_dual_mov_b32 v3, s7
-; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
; GFX12-GISEL-NEXT: buffer_store_b128 v[0:3], off, s[0:3], null
; GFX12-GISEL-NEXT: s_endpgm
%val = call <4 x float> @llvm.maxnum.v4f32(<4 x float> %a, <4 x float> %b)
@@ -664,54 +607,21 @@ define amdgpu_kernel void @test_fmax_v8f32(ptr addrspace(1) %out, <8 x float> %a
; GFX12-GISEL-NEXT: s_clause 0x1
; GFX12-GISEL-NEXT: s_load_b512 s[8:23], s[4:5], 0x44
; GFX12-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
+; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s8, s8
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s16, s16
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s9, s9
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s17, s17
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s10, s10
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v5, s18, s18
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v6, s11, s11
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v7, s19, s19
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v8, s12, s12
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v9, s20, s20
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v10, s13, s13
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v11, s21, s21
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v12, s14, s14
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v13, s22, s22
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v14, s15, s15
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v15, s23, s23
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s5, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s6, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s7, v4
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s8, v5
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s9, v6
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s10, v7
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s11, v8
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s12, v9
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s13, v10
-; GFX12-GISEL-NEXT: s_max_num_f32 s4, s2, s3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v11
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v12
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s14, v13
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s15, v14
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s16, v15
-; GFX12-GISEL-NEXT: s_max_num_f32 s5, s5, s6
-; GFX12-GISEL-NEXT: s_max_num_f32 s6, s7, s8
-; GFX12-GISEL-NEXT: s_max_num_f32 s7, s9, s10
-; GFX12-GISEL-NEXT: s_max_num_f32 s8, s11, s12
-; GFX12-GISEL-NEXT: s_max_num_f32 s9, s13, s2
-; GFX12-GISEL-NEXT: s_max_num_f32 s10, s3, s14
-; GFX12-GISEL-NEXT: s_max_num_f32 s11, s15, s16
+; GFX12-GISEL-NEXT: s_max_num_f32 s4, s8, s16
+; GFX12-GISEL-NEXT: s_max_num_f32 s5, s9, s17
+; GFX12-GISEL-NEXT: s_max_num_f32 s6, s10, s18
+; GFX12-GISEL-NEXT: s_max_num_f32 s7, s11, s19
+; GFX12-GISEL-NEXT: s_max_num_f32 s8, s12, s20
+; GFX12-GISEL-NEXT: s_max_num_f32 s9, s13, s21
+; GFX12-GISEL-NEXT: s_max_num_f32 s10, s14, s22
+; GFX12-GISEL-NEXT: s_max_num_f32 s11, s15, s23
; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX12-GISEL-NEXT: v_dual_mov_b32 v2, s6 :: v_dual_mov_b32 v3, s7
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: v_dual_mov_b32 v4, s8 :: v_dual_mov_b32 v5, s9
; GFX12-GISEL-NEXT: v_dual_mov_b32 v6, s10 :: v_dual_mov_b32 v7, s11
-; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
; GFX12-GISEL-NEXT: s_clause 0x1
; GFX12-GISEL-NEXT: buffer_store_b128 v[0:3], off, s[0:3], null
; GFX12-GISEL-NEXT: buffer_store_b128 v[4:7], off, s[0:3], null offset:16
@@ -1014,99 +924,34 @@ define amdgpu_kernel void @test_fmax_v16f32(ptr addrspace(1) %out, <16 x float>
; GFX12-GISEL-LABEL: test_fmax_v16f32:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_clause 0x2
-; GFX12-GISEL-NEXT: s_load_b512 s[36:51], s[4:5], 0x64
-; GFX12-GISEL-NEXT: s_load_b512 s[8:23], s[4:5], 0xa4
+; GFX12-GISEL-NEXT: s_load_b512 s[8:23], s[4:5], 0x64
+; GFX12-GISEL-NEXT: s_load_b512 s[36:51], s[4:5], 0xa4
; GFX12-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
+; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s36, s36
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s8, s8
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s37, s37
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s9, s9
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s38, s38
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v5, s10, s10
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v6, s39, s39
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v7, s11, s11
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v8, s40, s40
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v9, s12, s12
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v11, s13, s13
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s4, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s5, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s6, v4
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s7, v5
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s11, v6
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s12, v7
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s13, v8
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s24, v9
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s42, s42
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s14, s14
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s43, s43
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s15, s15
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s44, s44
-; GFX12-GISEL-NEXT: s_max_num_f32 s8, s2, s3
-; GFX12-GISEL-NEXT: s_max_num_f32 s9, s4, s5
-; GFX12-GISEL-NEXT: s_max_num_f32 s10, s6, s7
-; GFX12-GISEL-NEXT: s_max_num_f32 s11, s11, s12
-; GFX12-GISEL-NEXT: s_max_num_f32 s4, s13, s24
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s7, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s12, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s13, v4
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s16, s16
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s45, s45
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s17, s17
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s46, s46
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s18, s18
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s14, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s15, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s16, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s17, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s18, v4
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s47, s47
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s19, s19
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s48, s48
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s20, s20
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s49, s49
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v10, s41, s41
-; GFX12-GISEL-NEXT: s_max_num_f32 s6, s2, s3
-; GFX12-GISEL-NEXT: s_max_num_f32 s7, s7, s12
-; GFX12-GISEL-NEXT: s_max_num_f32 s12, s13, s14
-; GFX12-GISEL-NEXT: s_max_num_f32 s13, s15, s16
-; GFX12-GISEL-NEXT: s_max_num_f32 s14, s17, s18
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s16, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s17, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s18, v4
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s21, s21
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s50, s50
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s22, s22
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s51, s51
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s23, s23
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s25, v10
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s26, v11
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s19, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s20, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s21, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s22, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s23, v4
-; GFX12-GISEL-NEXT: s_max_num_f32 s5, s25, s26
-; GFX12-GISEL-NEXT: s_max_num_f32 s15, s2, s3
-; GFX12-GISEL-NEXT: s_max_num_f32 s16, s16, s17
-; GFX12-GISEL-NEXT: s_max_num_f32 s17, s18, s19
-; GFX12-GISEL-NEXT: s_max_num_f32 s18, s20, s21
-; GFX12-GISEL-NEXT: s_max_num_f32 s19, s22, s23
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s8 :: v_dual_mov_b32 v1, s9
-; GFX12-GISEL-NEXT: v_dual_mov_b32 v2, s10 :: v_dual_mov_b32 v3, s11
-; GFX12-GISEL-NEXT: v_dual_mov_b32 v4, s4 :: v_dual_mov_b32 v5, s5
-; GFX12-GISEL-NEXT: v_dual_mov_b32 v6, s6 :: v_dual_mov_b32 v7, s7
+; GFX12-GISEL-NEXT: s_max_num_f32 s4, s8, s36
+; GFX12-GISEL-NEXT: s_max_num_f32 s5, s9, s37
+; GFX12-GISEL-NEXT: s_max_num_f32 s6, s10, s38
+; GFX12-GISEL-NEXT: s_max_num_f32 s7, s11, s39
+; GFX12-GISEL-NEXT: s_max_num_f32 s8, s12, s40
+; GFX12-GISEL-NEXT: s_max_num_f32 s9, s13, s41
+; GFX12-GISEL-NEXT: s_max_num_f32 s10, s14, s42
+; GFX12-GISEL-NEXT: s_max_num_f32 s11, s15, s43
+; GFX12-GISEL-NEXT: s_max_num_f32 s12, s16, s44
+; GFX12-GISEL-NEXT: s_max_num_f32 s13, s17, s45
+; GFX12-GISEL-NEXT: s_max_num_f32 s14, s18, s46
+; GFX12-GISEL-NEXT: s_max_num_f32 s15, s19, s47
+; GFX12-GISEL-NEXT: s_max_num_f32 s16, s20, s48
+; GFX12-GISEL-NEXT: s_max_num_f32 s17, s21, s49
+; GFX12-GISEL-NEXT: s_max_num_f32 s18, s22, s50
+; GFX12-GISEL-NEXT: s_max_num_f32 s19, s23, s51
+; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
+; GFX12-GISEL-NEXT: v_dual_mov_b32 v2, s6 :: v_dual_mov_b32 v3, s7
+; GFX12-GISEL-NEXT: v_dual_mov_b32 v4, s8 :: v_dual_mov_b32 v5, s9
+; GFX12-GISEL-NEXT: v_dual_mov_b32 v6, s10 :: v_dual_mov_b32 v7, s11
; GFX12-GISEL-NEXT: v_dual_mov_b32 v8, s12 :: v_dual_mov_b32 v9, s13
; GFX12-GISEL-NEXT: v_dual_mov_b32 v10, s14 :: v_dual_mov_b32 v11, s15
-; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
; GFX12-GISEL-NEXT: v_dual_mov_b32 v12, s16 :: v_dual_mov_b32 v13, s17
; GFX12-GISEL-NEXT: v_dual_mov_b32 v14, s18 :: v_dual_mov_b32 v15, s19
; GFX12-GISEL-NEXT: s_clause 0x3
@@ -1785,31 +1630,16 @@ define <3 x float> @test_func_fmax_v3f32(<3 x float> %a, <3 x float> %b) #0 {
; GFX9-GISEL-NEXT: v_max_f32_e32 v2, v2, v3
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX12-SDAG-LABEL: test_func_fmax_v3f32:
-; GFX12-SDAG: ; %bb.0:
-; GFX12-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-SDAG-NEXT: s_wait_expcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_samplecnt 0x0
-; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: v_dual_max_num_f32 v0, v0, v3 :: v_dual_max_num_f32 v1, v1, v4
-; GFX12-SDAG-NEXT: v_max_num_f32_e32 v2, v2, v5
-; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-GISEL-LABEL: test_func_fmax_v3f32:
-; GFX12-GISEL: ; %bb.0:
-; GFX12-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-GISEL-NEXT: s_wait_expcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
-; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v3, v3, v3
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v1, v1, v1 :: v_dual_max_num_f32 v4, v4, v4
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v5, v5, v5
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v3 :: v_dual_max_num_f32 v1, v1, v4
-; GFX12-GISEL-NEXT: v_max_num_f32_e32 v2, v2, v5
-; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
+; GFX12-LABEL: test_func_fmax_v3f32:
+; GFX12: ; %bb.0:
+; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT: s_wait_expcnt 0x0
+; GFX12-NEXT: s_wait_samplecnt 0x0
+; GFX12-NEXT: s_wait_bvhcnt 0x0
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: v_dual_max_num_f32 v0, v0, v3 :: v_dual_max_num_f32 v1, v1, v4
+; GFX12-NEXT: v_max_num_f32_e32 v2, v2, v5
+; GFX12-NEXT: s_setpc_b64 s[30:31]
%val = call <3 x float> @llvm.maxnum.v3f32(<3 x float> %a, <3 x float> %b)
ret <3 x float> %val
}
@@ -1871,37 +1701,18 @@ define amdgpu_kernel void @test_fmax_f16_v_ieee_on(ptr addrspace(1) %out, half %
; GFX9-GISEL-NEXT: buffer_store_short v0, off, s[0:3], 0
; GFX9-GISEL-NEXT: s_endpgm
;
-; GFX12-SDAG-LABEL: test_fmax_f16_v_ieee_on:
-; GFX12-SDAG: ; %bb.0:
-; GFX12-SDAG-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: s_lshr_b32 s3, s2, 16
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
-; GFX12-SDAG-NEXT: s_max_num_f16 s2, s2, s3
-; GFX12-SDAG-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-SDAG-NEXT: v_mov_b16_e32 v0.l, s2
-; GFX12-SDAG-NEXT: s_mov_b32 s2, -1
-; GFX12-SDAG-NEXT: buffer_store_b16 v0, off, s[0:3], null
-; GFX12-SDAG-NEXT: s_endpgm
-;
-; GFX12-GISEL-LABEL: test_fmax_f16_v_ieee_on:
-; GFX12-GISEL: ; %bb.0:
-; GFX12-GISEL-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: s_lshr_b32 s3, s2, 16
-; GFX12-GISEL-NEXT: v_max_num_f16_e64 v0.l, s2, s2
-; GFX12-GISEL-NEXT: v_max_num_f16_e64 v1.l, s3, s3
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: s_max_num_f16 s2, s2, s3
-; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-NEXT: v_mov_b16_e32 v0.l, s2
-; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: buffer_store_b16 v0, off, s[0:3], null
-; GFX12-GISEL-NEXT: s_endpgm
+; GFX12-LABEL: test_fmax_f16_v_ieee_on:
+; GFX12: ; %bb.0:
+; GFX12-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: s_lshr_b32 s3, s2, 16
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
+; GFX12-NEXT: s_max_num_f16 s2, s2, s3
+; GFX12-NEXT: s_mov_b32 s3, 0x31016000
+; GFX12-NEXT: v_mov_b16_e32 v0.l, s2
+; GFX12-NEXT: s_mov_b32 s2, -1
+; GFX12-NEXT: buffer_store_b16 v0, off, s[0:3], null
+; GFX12-NEXT: s_endpgm
%val = call half @llvm.maxnum.f16(half %a, half %b)
store half %val, ptr addrspace(1) %out, align 2
ret void
@@ -2009,15 +1820,9 @@ define amdgpu_kernel void @test_fmax_f16_s_ieee_on(ptr addrspace(1) %out, half i
; GFX12-GISEL-NEXT: s_load_u16 s3, s[4:5], 0x2e
; GFX12-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f16_e64 v0.l, s2, s2
-; GFX12-GISEL-NEXT: v_max_num_f16_e64 v1.l, s3, s3
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
; GFX12-GISEL-NEXT: s_max_num_f16 s2, s2, s3
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: v_mov_b16_e32 v0.l, s2
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
; GFX12-GISEL-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -2102,35 +1907,17 @@ define amdgpu_kernel void @test_fmax_f32_s_ieee_on(ptr addrspace(1) %out, float
; GFX9-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; GFX9-GISEL-NEXT: s_endpgm
;
-; GFX12-SDAG-LABEL: test_fmax_f32_s_ieee_on:
-; GFX12-SDAG: ; %bb.0:
-; GFX12-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: s_max_num_f32 s2, s2, s3
-; GFX12-SDAG-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
-; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, s2
-; GFX12-SDAG-NEXT: s_mov_b32 s2, -1
-; GFX12-SDAG-NEXT: buffer_store_b32 v0, off, s[0:3], null
-; GFX12-SDAG-NEXT: s_endpgm
-;
-; GFX12-GISEL-LABEL: test_fmax_f32_s_ieee_on:
-; GFX12-GISEL: ; %bb.0:
-; GFX12-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s2, s2
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s3, s3
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: s_max_num_f32 s2, s2, s3
-; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-NEXT: v_mov_b32_e32 v0, s2
-; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: buffer_store_b32 v0, off, s[0:3], null
-; GFX12-GISEL-NEXT: s_endpgm
+; GFX12-LABEL: test_fmax_f32_s_ieee_on:
+; GFX12: ; %bb.0:
+; GFX12-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: s_max_num_f32 s2, s2, s3
+; GFX12-NEXT: s_mov_b32 s3, 0x31016000
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX12-NEXT: v_mov_b32_e32 v0, s2
+; GFX12-NEXT: s_mov_b32 s2, -1
+; GFX12-NEXT: buffer_store_b32 v0, off, s[0:3], null
+; GFX12-NEXT: s_endpgm
%val = call float @llvm.maxnum.f32(float %a, float %b)
store float %val, ptr addrspace(1) %out, align 4
ret void
@@ -2244,12 +2031,9 @@ define amdgpu_kernel void @test_fmax_v2f16_v_ieee_on(ptr addrspace(1) %out, <2 x
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_pk_max_num_f16 v0, s2, s2
-; GFX12-GISEL-NEXT: v_pk_max_num_f16 v1, s3, s3
+; GFX12-GISEL-NEXT: v_pk_max_num_f16 v0, s2, s3
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_pk_max_num_f16 v0, v0, v1
; GFX12-GISEL-NEXT: buffer_store_b32 v0, off, s[0:3], null
; GFX12-GISEL-NEXT: s_endpgm
%val = call <2 x half> @llvm.maxnum.v2f16(<2 x half> %a, <2 x half> %b)
@@ -2374,12 +2158,9 @@ define amdgpu_kernel void @test_fmax_v2f16_s_ieee_on(ptr addrspace(1) %out, <2 x
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_pk_max_num_f16 v0, s2, s2
-; GFX12-GISEL-NEXT: v_pk_max_num_f16 v1, s3, s3
+; GFX12-GISEL-NEXT: v_pk_max_num_f16 v0, s2, s3
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_pk_max_num_f16 v0, v0, v1
; GFX12-GISEL-NEXT: buffer_store_b32 v0, off, s[0:3], null
; GFX12-GISEL-NEXT: s_endpgm
%val = call <2 x half> @llvm.maxnum.v2f16(<2 x half> %a, <2 x half> %b)
@@ -2502,12 +2283,9 @@ define amdgpu_kernel void @test_fmax_f64_v_ieee_on(ptr addrspace(1) %out, double
; GFX12-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX12-GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e64 v[0:1], s[2:3], s[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e64 v[2:3], s[4:5], s[4:5]
+; GFX12-GISEL-NEXT: v_max_num_f64_e64 v[0:1], s[2:3], s[4:5]
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: buffer_store_b64 v[0:1], off, s[0:3], null
; GFX12-GISEL-NEXT: s_endpgm
%val = call double @llvm.maxnum.f64(double %a, double %b)
@@ -2612,12 +2390,9 @@ define amdgpu_kernel void @test_fmax_f64_s_ieee_on(ptr addrspace(1) %out, double
; GFX12-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX12-GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e64 v[0:1], s[2:3], s[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e64 v[2:3], s[4:5], s[4:5]
+; GFX12-GISEL-NEXT: v_max_num_f64_e64 v[0:1], s[2:3], s[4:5]
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: buffer_store_b64 v[0:1], off, s[0:3], null
; GFX12-GISEL-NEXT: s_endpgm
%val = call double @llvm.maxnum.f64(double %a, double %b)
diff --git a/llvm/test/CodeGen/AMDGPU/fminnum.ll b/llvm/test/CodeGen/AMDGPU/fminnum.ll
index 1e308160adb0a..3d3e08895e83a 100644
--- a/llvm/test/CodeGen/AMDGPU/fminnum.ll
+++ b/llvm/test/CodeGen/AMDGPU/fminnum.ll
@@ -58,35 +58,17 @@ define amdgpu_kernel void @test_fmin_f32_ieee_mode_on(ptr addrspace(1) %out, flo
; GFX9-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; GFX9-GISEL-NEXT: s_endpgm
;
-; GFX12-SDAG-LABEL: test_fmin_f32_ieee_mode_on:
-; GFX12-SDAG: ; %bb.0:
-; GFX12-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: s_min_num_f32 s2, s2, s3
-; GFX12-SDAG-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
-; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, s2
-; GFX12-SDAG-NEXT: s_mov_b32 s2, -1
-; GFX12-SDAG-NEXT: buffer_store_b32 v0, off, s[0:3], null
-; GFX12-SDAG-NEXT: s_endpgm
-;
-; GFX12-GISEL-LABEL: test_fmin_f32_ieee_mode_on:
-; GFX12-GISEL: ; %bb.0:
-; GFX12-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s2, s2
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s3, s3
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: s_min_num_f32 s2, s2, s3
-; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-NEXT: v_mov_b32_e32 v0, s2
-; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: buffer_store_b32 v0, off, s[0:3], null
-; GFX12-GISEL-NEXT: s_endpgm
+; GFX12-LABEL: test_fmin_f32_ieee_mode_on:
+; GFX12: ; %bb.0:
+; GFX12-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: s_min_num_f32 s2, s2, s3
+; GFX12-NEXT: s_mov_b32 s3, 0x31016000
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX12-NEXT: v_mov_b32_e32 v0, s2
+; GFX12-NEXT: s_mov_b32 s2, -1
+; GFX12-NEXT: buffer_store_b32 v0, off, s[0:3], null
+; GFX12-NEXT: s_endpgm
%val = call float @llvm.minnum.f32(float %a, float %b)
store float %val, ptr addrspace(1) %out, align 4
ret void
@@ -244,20 +226,9 @@ define amdgpu_kernel void @test_fmin_v2f32(ptr addrspace(1) %out, <2 x float> %a
; GFX12-GISEL-NEXT: s_mov_b32 s6, -1
; GFX12-GISEL-NEXT: s_mov_b32 s7, 0x31016000
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s0, s0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s2, s2
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s1, s1
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s3, s3
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s0, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s1, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v3
-; GFX12-GISEL-NEXT: s_min_num_f32 s0, s0, s1
-; GFX12-GISEL-NEXT: s_min_num_f32 s1, s2, s3
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX12-GISEL-NEXT: s_min_num_f32 s0, s0, s2
+; GFX12-GISEL-NEXT: s_min_num_f32 s1, s1, s3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-GISEL-NEXT: buffer_store_b64 v[0:1], off, s[4:7], null
; GFX12-GISEL-NEXT: s_endpgm
@@ -378,32 +349,16 @@ define amdgpu_kernel void @test_fmin_v4f32(ptr addrspace(1) %out, <4 x float> %a
; GFX12-GISEL-NEXT: s_clause 0x1
; GFX12-GISEL-NEXT: s_load_b256 s[8:15], s[4:5], 0x34
; GFX12-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s8, s8
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s12, s12
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s9, s9
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s13, s13
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s10, s10
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v5, s14, s14
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v6, s11, s11
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v7, s15, s15
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s5, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s6, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s7, v4
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s8, v5
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s9, v6
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s10, v7
-; GFX12-GISEL-NEXT: s_min_num_f32 s4, s2, s3
-; GFX12-GISEL-NEXT: s_min_num_f32 s5, s5, s6
-; GFX12-GISEL-NEXT: s_min_num_f32 s6, s7, s8
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: s_min_num_f32 s7, s9, s10
+; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
+; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT: s_min_num_f32 s4, s8, s12
+; GFX12-GISEL-NEXT: s_min_num_f32 s5, s9, s13
+; GFX12-GISEL-NEXT: s_min_num_f32 s6, s10, s14
+; GFX12-GISEL-NEXT: s_min_num_f32 s7, s11, s15
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: v_dual_mov_b32 v2, s6 :: v_dual_mov_b32 v3, s7
-; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
; GFX12-GISEL-NEXT: buffer_store_b128 v[0:3], off, s[0:3], null
; GFX12-GISEL-NEXT: s_endpgm
%val = call <4 x float> @llvm.minnum.v4f32(<4 x float> %a, <4 x float> %b)
@@ -582,54 +537,21 @@ define amdgpu_kernel void @test_fmin_v8f32(ptr addrspace(1) %out, <8 x float> %a
; GFX12-GISEL-NEXT: s_clause 0x1
; GFX12-GISEL-NEXT: s_load_b512 s[8:23], s[4:5], 0x44
; GFX12-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
+; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s8, s8
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s16, s16
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s9, s9
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s17, s17
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s10, s10
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v5, s18, s18
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v6, s11, s11
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v7, s19, s19
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v8, s12, s12
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v9, s20, s20
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v10, s13, s13
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v11, s21, s21
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v12, s14, s14
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v13, s22, s22
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v14, s15, s15
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v15, s23, s23
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s5, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s6, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s7, v4
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s8, v5
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s9, v6
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s10, v7
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s11, v8
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s12, v9
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s13, v10
-; GFX12-GISEL-NEXT: s_min_num_f32 s4, s2, s3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v11
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v12
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s14, v13
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s15, v14
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s16, v15
-; GFX12-GISEL-NEXT: s_min_num_f32 s5, s5, s6
-; GFX12-GISEL-NEXT: s_min_num_f32 s6, s7, s8
-; GFX12-GISEL-NEXT: s_min_num_f32 s7, s9, s10
-; GFX12-GISEL-NEXT: s_min_num_f32 s8, s11, s12
-; GFX12-GISEL-NEXT: s_min_num_f32 s9, s13, s2
-; GFX12-GISEL-NEXT: s_min_num_f32 s10, s3, s14
-; GFX12-GISEL-NEXT: s_min_num_f32 s11, s15, s16
+; GFX12-GISEL-NEXT: s_min_num_f32 s4, s8, s16
+; GFX12-GISEL-NEXT: s_min_num_f32 s5, s9, s17
+; GFX12-GISEL-NEXT: s_min_num_f32 s6, s10, s18
+; GFX12-GISEL-NEXT: s_min_num_f32 s7, s11, s19
+; GFX12-GISEL-NEXT: s_min_num_f32 s8, s12, s20
+; GFX12-GISEL-NEXT: s_min_num_f32 s9, s13, s21
+; GFX12-GISEL-NEXT: s_min_num_f32 s10, s14, s22
+; GFX12-GISEL-NEXT: s_min_num_f32 s11, s15, s23
; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX12-GISEL-NEXT: v_dual_mov_b32 v2, s6 :: v_dual_mov_b32 v3, s7
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: v_dual_mov_b32 v4, s8 :: v_dual_mov_b32 v5, s9
; GFX12-GISEL-NEXT: v_dual_mov_b32 v6, s10 :: v_dual_mov_b32 v7, s11
-; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
; GFX12-GISEL-NEXT: s_clause 0x1
; GFX12-GISEL-NEXT: buffer_store_b128 v[0:3], off, s[0:3], null
; GFX12-GISEL-NEXT: buffer_store_b128 v[4:7], off, s[0:3], null offset:16
@@ -932,99 +854,34 @@ define amdgpu_kernel void @test_fmin_v16f32(ptr addrspace(1) %out, <16 x float>
; GFX12-GISEL-LABEL: test_fmin_v16f32:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_clause 0x2
-; GFX12-GISEL-NEXT: s_load_b512 s[36:51], s[4:5], 0x64
-; GFX12-GISEL-NEXT: s_load_b512 s[8:23], s[4:5], 0xa4
+; GFX12-GISEL-NEXT: s_load_b512 s[8:23], s[4:5], 0x64
+; GFX12-GISEL-NEXT: s_load_b512 s[36:51], s[4:5], 0xa4
; GFX12-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
+; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s36, s36
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s8, s8
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s37, s37
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s9, s9
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s38, s38
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v5, s10, s10
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v6, s39, s39
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v7, s11, s11
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v8, s40, s40
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v9, s12, s12
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v11, s13, s13
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s4, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s5, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s6, v4
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s7, v5
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s11, v6
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s12, v7
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s13, v8
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s24, v9
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s42, s42
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s14, s14
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s43, s43
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s15, s15
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s44, s44
-; GFX12-GISEL-NEXT: s_min_num_f32 s8, s2, s3
-; GFX12-GISEL-NEXT: s_min_num_f32 s9, s4, s5
-; GFX12-GISEL-NEXT: s_min_num_f32 s10, s6, s7
-; GFX12-GISEL-NEXT: s_min_num_f32 s11, s11, s12
-; GFX12-GISEL-NEXT: s_min_num_f32 s4, s13, s24
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s7, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s12, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s13, v4
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s16, s16
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s45, s45
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s17, s17
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s46, s46
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s18, s18
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s14, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s15, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s16, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s17, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s18, v4
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s47, s47
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s19, s19
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s48, s48
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s20, s20
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s49, s49
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v10, s41, s41
-; GFX12-GISEL-NEXT: s_min_num_f32 s6, s2, s3
-; GFX12-GISEL-NEXT: s_min_num_f32 s7, s7, s12
-; GFX12-GISEL-NEXT: s_min_num_f32 s12, s13, s14
-; GFX12-GISEL-NEXT: s_min_num_f32 s13, s15, s16
-; GFX12-GISEL-NEXT: s_min_num_f32 s14, s17, s18
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s16, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s17, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s18, v4
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s21, s21
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s50, s50
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v2, s22, s22
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v3, s51, s51
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v4, s23, s23
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s25, v10
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s26, v11
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s19, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s20, v1
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s21, v2
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s22, v3
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s23, v4
-; GFX12-GISEL-NEXT: s_min_num_f32 s5, s25, s26
-; GFX12-GISEL-NEXT: s_min_num_f32 s15, s2, s3
-; GFX12-GISEL-NEXT: s_min_num_f32 s16, s16, s17
-; GFX12-GISEL-NEXT: s_min_num_f32 s17, s18, s19
-; GFX12-GISEL-NEXT: s_min_num_f32 s18, s20, s21
-; GFX12-GISEL-NEXT: s_min_num_f32 s19, s22, s23
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s8 :: v_dual_mov_b32 v1, s9
-; GFX12-GISEL-NEXT: v_dual_mov_b32 v2, s10 :: v_dual_mov_b32 v3, s11
-; GFX12-GISEL-NEXT: v_dual_mov_b32 v4, s4 :: v_dual_mov_b32 v5, s5
-; GFX12-GISEL-NEXT: v_dual_mov_b32 v6, s6 :: v_dual_mov_b32 v7, s7
+; GFX12-GISEL-NEXT: s_min_num_f32 s4, s8, s36
+; GFX12-GISEL-NEXT: s_min_num_f32 s5, s9, s37
+; GFX12-GISEL-NEXT: s_min_num_f32 s6, s10, s38
+; GFX12-GISEL-NEXT: s_min_num_f32 s7, s11, s39
+; GFX12-GISEL-NEXT: s_min_num_f32 s8, s12, s40
+; GFX12-GISEL-NEXT: s_min_num_f32 s9, s13, s41
+; GFX12-GISEL-NEXT: s_min_num_f32 s10, s14, s42
+; GFX12-GISEL-NEXT: s_min_num_f32 s11, s15, s43
+; GFX12-GISEL-NEXT: s_min_num_f32 s12, s16, s44
+; GFX12-GISEL-NEXT: s_min_num_f32 s13, s17, s45
+; GFX12-GISEL-NEXT: s_min_num_f32 s14, s18, s46
+; GFX12-GISEL-NEXT: s_min_num_f32 s15, s19, s47
+; GFX12-GISEL-NEXT: s_min_num_f32 s16, s20, s48
+; GFX12-GISEL-NEXT: s_min_num_f32 s17, s21, s49
+; GFX12-GISEL-NEXT: s_min_num_f32 s18, s22, s50
+; GFX12-GISEL-NEXT: s_min_num_f32 s19, s23, s51
+; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
+; GFX12-GISEL-NEXT: v_dual_mov_b32 v2, s6 :: v_dual_mov_b32 v3, s7
+; GFX12-GISEL-NEXT: v_dual_mov_b32 v4, s8 :: v_dual_mov_b32 v5, s9
+; GFX12-GISEL-NEXT: v_dual_mov_b32 v6, s10 :: v_dual_mov_b32 v7, s11
; GFX12-GISEL-NEXT: v_dual_mov_b32 v8, s12 :: v_dual_mov_b32 v9, s13
; GFX12-GISEL-NEXT: v_dual_mov_b32 v10, s14 :: v_dual_mov_b32 v11, s15
-; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
; GFX12-GISEL-NEXT: v_dual_mov_b32 v12, s16 :: v_dual_mov_b32 v13, s17
; GFX12-GISEL-NEXT: v_dual_mov_b32 v14, s18 :: v_dual_mov_b32 v15, s19
; GFX12-GISEL-NEXT: s_clause 0x3
@@ -1701,31 +1558,16 @@ define <3 x float> @test_func_fmin_v3f32(<3 x float> %a, <3 x float> %b) nounwin
; GFX9-GISEL-NEXT: v_min_f32_e32 v2, v2, v3
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX12-SDAG-LABEL: test_func_fmin_v3f32:
-; GFX12-SDAG: ; %bb.0:
-; GFX12-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-SDAG-NEXT: s_wait_expcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_samplecnt 0x0
-; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: v_dual_min_num_f32 v0, v0, v3 :: v_dual_min_num_f32 v1, v1, v4
-; GFX12-SDAG-NEXT: v_min_num_f32_e32 v2, v2, v5
-; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-GISEL-LABEL: test_func_fmin_v3f32:
-; GFX12-GISEL: ; %bb.0:
-; GFX12-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-GISEL-NEXT: s_wait_expcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
-; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v3, v3, v3
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v1, v1, v1 :: v_dual_max_num_f32 v4, v4, v4
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v5, v5, v5
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-NEXT: v_dual_min_num_f32 v0, v0, v3 :: v_dual_min_num_f32 v1, v1, v4
-; GFX12-GISEL-NEXT: v_min_num_f32_e32 v2, v2, v5
-; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
+; GFX12-LABEL: test_func_fmin_v3f32:
+; GFX12: ; %bb.0:
+; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT: s_wait_expcnt 0x0
+; GFX12-NEXT: s_wait_samplecnt 0x0
+; GFX12-NEXT: s_wait_bvhcnt 0x0
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: v_dual_min_num_f32 v0, v0, v3 :: v_dual_min_num_f32 v1, v1, v4
+; GFX12-NEXT: v_min_num_f32_e32 v2, v2, v5
+; GFX12-NEXT: s_setpc_b64 s[30:31]
%val = call <3 x float> @llvm.minnum.v3f32(<3 x float> %a, <3 x float> %b)
ret <3 x float> %val
}
@@ -1787,37 +1629,18 @@ define amdgpu_kernel void @test_fmin_f16_v_ieee_on(ptr addrspace(1) %out, half %
; GFX9-GISEL-NEXT: buffer_store_short v0, off, s[0:3], 0
; GFX9-GISEL-NEXT: s_endpgm
;
-; GFX12-SDAG-LABEL: test_fmin_f16_v_ieee_on:
-; GFX12-SDAG: ; %bb.0:
-; GFX12-SDAG-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: s_lshr_b32 s3, s2, 16
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
-; GFX12-SDAG-NEXT: s_min_num_f16 s2, s2, s3
-; GFX12-SDAG-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-SDAG-NEXT: v_mov_b16_e32 v0.l, s2
-; GFX12-SDAG-NEXT: s_mov_b32 s2, -1
-; GFX12-SDAG-NEXT: buffer_store_b16 v0, off, s[0:3], null
-; GFX12-SDAG-NEXT: s_endpgm
-;
-; GFX12-GISEL-LABEL: test_fmin_f16_v_ieee_on:
-; GFX12-GISEL: ; %bb.0:
-; GFX12-GISEL-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: s_lshr_b32 s3, s2, 16
-; GFX12-GISEL-NEXT: v_max_num_f16_e64 v0.l, s2, s2
-; GFX12-GISEL-NEXT: v_max_num_f16_e64 v1.l, s3, s3
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: s_min_num_f16 s2, s2, s3
-; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-NEXT: v_mov_b16_e32 v0.l, s2
-; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: buffer_store_b16 v0, off, s[0:3], null
-; GFX12-GISEL-NEXT: s_endpgm
+; GFX12-LABEL: test_fmin_f16_v_ieee_on:
+; GFX12: ; %bb.0:
+; GFX12-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: s_lshr_b32 s3, s2, 16
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
+; GFX12-NEXT: s_min_num_f16 s2, s2, s3
+; GFX12-NEXT: s_mov_b32 s3, 0x31016000
+; GFX12-NEXT: v_mov_b16_e32 v0.l, s2
+; GFX12-NEXT: s_mov_b32 s2, -1
+; GFX12-NEXT: buffer_store_b16 v0, off, s[0:3], null
+; GFX12-NEXT: s_endpgm
%val = call half @llvm.minnum.f16(half %a, half %b)
store half %val, ptr addrspace(1) %out, align 2
ret void
@@ -1925,15 +1748,9 @@ define amdgpu_kernel void @test_fmin_f16_s_ieee_on(ptr addrspace(1) %out, half i
; GFX12-GISEL-NEXT: s_load_u16 s3, s[4:5], 0x2e
; GFX12-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f16_e64 v0.l, s2, s2
-; GFX12-GISEL-NEXT: v_max_num_f16_e64 v1.l, s3, s3
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
; GFX12-GISEL-NEXT: s_min_num_f16 s2, s2, s3
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: v_mov_b16_e32 v0.l, s2
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
; GFX12-GISEL-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -2018,35 +1835,17 @@ define amdgpu_kernel void @test_fmin_f32_s_ieee_on(ptr addrspace(1) %out, float
; GFX9-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; GFX9-GISEL-NEXT: s_endpgm
;
-; GFX12-SDAG-LABEL: test_fmin_f32_s_ieee_on:
-; GFX12-SDAG: ; %bb.0:
-; GFX12-SDAG-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: s_min_num_f32 s2, s2, s3
-; GFX12-SDAG-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
-; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, s2
-; GFX12-SDAG-NEXT: s_mov_b32 s2, -1
-; GFX12-SDAG-NEXT: buffer_store_b32 v0, off, s[0:3], null
-; GFX12-SDAG-NEXT: s_endpgm
-;
-; GFX12-GISEL-LABEL: test_fmin_f32_s_ieee_on:
-; GFX12-GISEL: ; %bb.0:
-; GFX12-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v0, s2, s2
-; GFX12-GISEL-NEXT: v_max_num_f32_e64 v1, s3, s3
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
-; GFX12-GISEL-NEXT: s_min_num_f32 s2, s2, s3
-; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-NEXT: v_mov_b32_e32 v0, s2
-; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
-; GFX12-GISEL-NEXT: buffer_store_b32 v0, off, s[0:3], null
-; GFX12-GISEL-NEXT: s_endpgm
+; GFX12-LABEL: test_fmin_f32_s_ieee_on:
+; GFX12: ; %bb.0:
+; GFX12-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: s_min_num_f32 s2, s2, s3
+; GFX12-NEXT: s_mov_b32 s3, 0x31016000
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX12-NEXT: v_mov_b32_e32 v0, s2
+; GFX12-NEXT: s_mov_b32 s2, -1
+; GFX12-NEXT: buffer_store_b32 v0, off, s[0:3], null
+; GFX12-NEXT: s_endpgm
%val = call float @llvm.minnum.f32(float %a, float %b)
store float %val, ptr addrspace(1) %out, align 4
ret void
@@ -2160,12 +1959,9 @@ define amdgpu_kernel void @test_fmin_v2f16_v_ieee_on(ptr addrspace(1) %out, <2 x
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_pk_max_num_f16 v0, s2, s2
-; GFX12-GISEL-NEXT: v_pk_max_num_f16 v1, s3, s3
+; GFX12-GISEL-NEXT: v_pk_min_num_f16 v0, s2, s3
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_pk_min_num_f16 v0, v0, v1
; GFX12-GISEL-NEXT: buffer_store_b32 v0, off, s[0:3], null
; GFX12-GISEL-NEXT: s_endpgm
%val = call <2 x half> @llvm.minnum.v2f16(<2 x half> %a, <2 x half> %b)
@@ -2290,12 +2086,9 @@ define amdgpu_kernel void @test_fmin_v2f16_s_ieee_on(ptr addrspace(1) %out, <2 x
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_pk_max_num_f16 v0, s2, s2
-; GFX12-GISEL-NEXT: v_pk_max_num_f16 v1, s3, s3
+; GFX12-GISEL-NEXT: v_pk_min_num_f16 v0, s2, s3
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_pk_min_num_f16 v0, v0, v1
; GFX12-GISEL-NEXT: buffer_store_b32 v0, off, s[0:3], null
; GFX12-GISEL-NEXT: s_endpgm
%val = call <2 x half> @llvm.minnum.v2f16(<2 x half> %a, <2 x half> %b)
@@ -2418,12 +2211,9 @@ define amdgpu_kernel void @test_fmin_f64_v_ieee_on(ptr addrspace(1) %out, double
; GFX12-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX12-GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e64 v[0:1], s[2:3], s[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e64 v[2:3], s[4:5], s[4:5]
+; GFX12-GISEL-NEXT: v_min_num_f64_e64 v[0:1], s[2:3], s[4:5]
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: buffer_store_b64 v[0:1], off, s[0:3], null
; GFX12-GISEL-NEXT: s_endpgm
%val = call double @llvm.minnum.f64(double %a, double %b)
@@ -2528,12 +2318,9 @@ define amdgpu_kernel void @test_fmin_f64_s_ieee_on(ptr addrspace(1) %out, double
; GFX12-GISEL-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX12-GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e64 v[0:1], s[2:3], s[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e64 v[2:3], s[4:5], s[4:5]
+; GFX12-GISEL-NEXT: v_min_num_f64_e64 v[0:1], s[2:3], s[4:5]
; GFX12-GISEL-NEXT: s_mov_b32 s2, -1
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0x31016000
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: buffer_store_b64 v[0:1], off, s[0:3], null
; GFX12-GISEL-NEXT: s_endpgm
%val = call double @llvm.minnum.f64(double %a, double %b)
diff --git a/llvm/test/CodeGen/AMDGPU/minmax.ll b/llvm/test/CodeGen/AMDGPU/minmax.ll
index c3bf35ae4ed53..2b69df9d216d3 100644
--- a/llvm/test/CodeGen/AMDGPU/minmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/minmax.ll
@@ -575,57 +575,28 @@ define float @test_minmax_f32_ieee_true(float %a, float %b, float %c) {
; GISEL-GFX11-NEXT: v_maxmin_f32 v0, v0, v1, v2
; GISEL-GFX11-NEXT: s_setpc_b64 s[30:31]
;
-; SDAG-GFX1170-LABEL: test_minmax_f32_ieee_true:
-; SDAG-GFX1170: ; %bb.0:
-; SDAG-GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; SDAG-GFX1170-NEXT: v_maxmin_num_f32 v0, v0, v1, v2
-; SDAG-GFX1170-NEXT: s_setpc_b64 s[30:31]
-;
-; GISEL-GFX1170-LABEL: test_minmax_f32_ieee_true:
-; GISEL-GFX1170: ; %bb.0:
-; GISEL-GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GISEL-GFX1170-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GISEL-GFX1170-NEXT: v_max_num_f32_e32 v2, v2, v2
-; GISEL-GFX1170-NEXT: v_maxmin_num_f32 v0, v0, v1, v2
-; GISEL-GFX1170-NEXT: s_setpc_b64 s[30:31]
+; GFX1170-LABEL: test_minmax_f32_ieee_true:
+; GFX1170: ; %bb.0:
+; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1170-NEXT: v_maxmin_num_f32 v0, v0, v1, v2
+; GFX1170-NEXT: s_setpc_b64 s[30:31]
;
-; SDAG-GFX12-LABEL: test_minmax_f32_ieee_true:
-; SDAG-GFX12: ; %bb.0:
-; SDAG-GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
-; SDAG-GFX12-NEXT: s_wait_expcnt 0x0
-; SDAG-GFX12-NEXT: s_wait_samplecnt 0x0
-; SDAG-GFX12-NEXT: s_wait_bvhcnt 0x0
-; SDAG-GFX12-NEXT: s_wait_kmcnt 0x0
-; SDAG-GFX12-NEXT: v_maxmin_num_f32 v0, v0, v1, v2
-; SDAG-GFX12-NEXT: s_setpc_b64 s[30:31]
-;
-; GISEL-GFX12-LABEL: test_minmax_f32_ieee_true:
-; GISEL-GFX12: ; %bb.0:
-; GISEL-GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
-; GISEL-GFX12-NEXT: s_wait_expcnt 0x0
-; GISEL-GFX12-NEXT: s_wait_samplecnt 0x0
-; GISEL-GFX12-NEXT: s_wait_bvhcnt 0x0
-; GISEL-GFX12-NEXT: s_wait_kmcnt 0x0
-; GISEL-GFX12-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GISEL-GFX12-NEXT: v_max_num_f32_e32 v2, v2, v2
-; GISEL-GFX12-NEXT: v_maxmin_num_f32 v0, v0, v1, v2
-; GISEL-GFX12-NEXT: s_setpc_b64 s[30:31]
-;
-; SDAG-GFX1250-LABEL: test_minmax_f32_ieee_true:
-; SDAG-GFX1250: ; %bb.0:
-; SDAG-GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
-; SDAG-GFX1250-NEXT: s_wait_kmcnt 0x0
-; SDAG-GFX1250-NEXT: v_maxmin_num_f32 v0, v0, v1, v2
-; SDAG-GFX1250-NEXT: s_set_pc_i64 s[30:31]
+; GFX12-LABEL: test_minmax_f32_ieee_true:
+; GFX12: ; %bb.0:
+; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT: s_wait_expcnt 0x0
+; GFX12-NEXT: s_wait_samplecnt 0x0
+; GFX12-NEXT: s_wait_bvhcnt 0x0
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: v_maxmin_num_f32 v0, v0, v1, v2
+; GFX12-NEXT: s_setpc_b64 s[30:31]
;
-; GISEL-GFX1250-LABEL: test_minmax_f32_ieee_true:
-; GISEL-GFX1250: ; %bb.0:
-; GISEL-GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
-; GISEL-GFX1250-NEXT: s_wait_kmcnt 0x0
-; GISEL-GFX1250-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GISEL-GFX1250-NEXT: v_max_num_f32_e32 v2, v2, v2
-; GISEL-GFX1250-NEXT: v_maxmin_num_f32 v0, v0, v1, v2
-; GISEL-GFX1250-NEXT: s_set_pc_i64 s[30:31]
+; GFX1250-LABEL: test_minmax_f32_ieee_true:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: v_maxmin_num_f32 v0, v0, v1, v2
+; GFX1250-NEXT: s_set_pc_i64 s[30:31]
%max = call float @llvm.maxnum.f32(float %a, float %b)
%minmax = call float @llvm.minnum.f32(float %max, float %c)
ret float %minmax
@@ -769,57 +740,28 @@ define float @test_maxmin_f32_ieee_true(float %a, float %b, float %c) {
; GISEL-GFX11-NEXT: v_minmax_f32 v0, v0, v1, v2
; GISEL-GFX11-NEXT: s_setpc_b64 s[30:31]
;
-; SDAG-GFX1170-LABEL: test_maxmin_f32_ieee_true:
-; SDAG-GFX1170: ; %bb.0:
-; SDAG-GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; SDAG-GFX1170-NEXT: v_minmax_num_f32 v0, v0, v1, v2
-; SDAG-GFX1170-NEXT: s_setpc_b64 s[30:31]
-;
-; GISEL-GFX1170-LABEL: test_maxmin_f32_ieee_true:
-; GISEL-GFX1170: ; %bb.0:
-; GISEL-GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GISEL-GFX1170-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GISEL-GFX1170-NEXT: v_max_num_f32_e32 v2, v2, v2
-; GISEL-GFX1170-NEXT: v_minmax_num_f32 v0, v0, v1, v2
-; GISEL-GFX1170-NEXT: s_setpc_b64 s[30:31]
+; GFX1170-LABEL: test_maxmin_f32_ieee_true:
+; GFX1170: ; %bb.0:
+; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1170-NEXT: v_minmax_num_f32 v0, v0, v1, v2
+; GFX1170-NEXT: s_setpc_b64 s[30:31]
;
-; SDAG-GFX12-LABEL: test_maxmin_f32_ieee_true:
-; SDAG-GFX12: ; %bb.0:
-; SDAG-GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
-; SDAG-GFX12-NEXT: s_wait_expcnt 0x0
-; SDAG-GFX12-NEXT: s_wait_samplecnt 0x0
-; SDAG-GFX12-NEXT: s_wait_bvhcnt 0x0
-; SDAG-GFX12-NEXT: s_wait_kmcnt 0x0
-; SDAG-GFX12-NEXT: v_minmax_num_f32 v0, v0, v1, v2
-; SDAG-GFX12-NEXT: s_setpc_b64 s[30:31]
-;
-; GISEL-GFX12-LABEL: test_maxmin_f32_ieee_true:
-; GISEL-GFX12: ; %bb.0:
-; GISEL-GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
-; GISEL-GFX12-NEXT: s_wait_expcnt 0x0
-; GISEL-GFX12-NEXT: s_wait_samplecnt 0x0
-; GISEL-GFX12-NEXT: s_wait_bvhcnt 0x0
-; GISEL-GFX12-NEXT: s_wait_kmcnt 0x0
-; GISEL-GFX12-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GISEL-GFX12-NEXT: v_max_num_f32_e32 v2, v2, v2
-; GISEL-GFX12-NEXT: v_minmax_num_f32 v0, v0, v1, v2
-; GISEL-GFX12-NEXT: s_setpc_b64 s[30:31]
-;
-; SDAG-GFX1250-LABEL: test_maxmin_f32_ieee_true:
-; SDAG-GFX1250: ; %bb.0:
-; SDAG-GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
-; SDAG-GFX1250-NEXT: s_wait_kmcnt 0x0
-; SDAG-GFX1250-NEXT: v_minmax_num_f32 v0, v0, v1, v2
-; SDAG-GFX1250-NEXT: s_set_pc_i64 s[30:31]
+; GFX12-LABEL: test_maxmin_f32_ieee_true:
+; GFX12: ; %bb.0:
+; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT: s_wait_expcnt 0x0
+; GFX12-NEXT: s_wait_samplecnt 0x0
+; GFX12-NEXT: s_wait_bvhcnt 0x0
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: v_minmax_num_f32 v0, v0, v1, v2
+; GFX12-NEXT: s_setpc_b64 s[30:31]
;
-; GISEL-GFX1250-LABEL: test_maxmin_f32_ieee_true:
-; GISEL-GFX1250: ; %bb.0:
-; GISEL-GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
-; GISEL-GFX1250-NEXT: s_wait_kmcnt 0x0
-; GISEL-GFX1250-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GISEL-GFX1250-NEXT: v_max_num_f32_e32 v2, v2, v2
-; GISEL-GFX1250-NEXT: v_minmax_num_f32 v0, v0, v1, v2
-; GISEL-GFX1250-NEXT: s_set_pc_i64 s[30:31]
+; GFX1250-LABEL: test_maxmin_f32_ieee_true:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: v_minmax_num_f32 v0, v0, v1, v2
+; GFX1250-NEXT: s_set_pc_i64 s[30:31]
%min = call float @llvm.minnum.f32(float %a, float %b)
%maxmin = call float @llvm.maxnum.f32(float %min, float %c)
ret float %maxmin
@@ -1272,18 +1214,12 @@ define half @test_minmax_commuted_f16_ieee_true(half %a, half %b, half %c) {
; GISEL-GFX1170-TRUE16-LABEL: test_minmax_commuted_f16_ieee_true:
; GISEL-GFX1170-TRUE16: ; %bb.0:
; GISEL-GFX1170-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GISEL-GFX1170-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GISEL-GFX1170-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v1.l, v1.l
-; GISEL-GFX1170-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v2.l, v2.l
-; GISEL-GFX1170-TRUE16-NEXT: v_maxmin_num_f16 v0.l, v0.l, v0.h, v1.l
+; GISEL-GFX1170-TRUE16-NEXT: v_maxmin_num_f16 v0.l, v0.l, v1.l, v2.l
; GISEL-GFX1170-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GISEL-GFX1170-FAKE16-LABEL: test_minmax_commuted_f16_ieee_true:
; GISEL-GFX1170-FAKE16: ; %bb.0:
; GISEL-GFX1170-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GISEL-GFX1170-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GISEL-GFX1170-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GISEL-GFX1170-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
; GISEL-GFX1170-FAKE16-NEXT: v_maxmin_num_f16 v0, v0, v1, v2
; GISEL-GFX1170-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1314,10 +1250,7 @@ define half @test_minmax_commuted_f16_ieee_true(half %a, half %b, half %c) {
; GISEL-GFX12-TRUE16-NEXT: s_wait_samplecnt 0x0
; GISEL-GFX12-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GISEL-GFX12-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GISEL-GFX12-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GISEL-GFX12-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v1.l, v1.l
-; GISEL-GFX12-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v2.l, v2.l
-; GISEL-GFX12-TRUE16-NEXT: v_maxmin_num_f16 v0.l, v0.l, v0.h, v1.l
+; GISEL-GFX12-TRUE16-NEXT: v_maxmin_num_f16 v0.l, v0.l, v1.l, v2.l
; GISEL-GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GISEL-GFX12-FAKE16-LABEL: test_minmax_commuted_f16_ieee_true:
@@ -1327,9 +1260,6 @@ define half @test_minmax_commuted_f16_ieee_true(half %a, half %b, half %c) {
; GISEL-GFX12-FAKE16-NEXT: s_wait_samplecnt 0x0
; GISEL-GFX12-FAKE16-NEXT: s_wait_bvhcnt 0x0
; GISEL-GFX12-FAKE16-NEXT: s_wait_kmcnt 0x0
-; GISEL-GFX12-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GISEL-GFX12-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GISEL-GFX12-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
; GISEL-GFX12-FAKE16-NEXT: v_maxmin_num_f16 v0, v0, v1, v2
; GISEL-GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1351,19 +1281,13 @@ define half @test_minmax_commuted_f16_ieee_true(half %a, half %b, half %c) {
; GISEL-GFX1250-TRUE16: ; %bb.0:
; GISEL-GFX1250-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
; GISEL-GFX1250-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GISEL-GFX1250-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GISEL-GFX1250-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v1.l, v1.l
-; GISEL-GFX1250-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v2.l, v2.l
-; GISEL-GFX1250-TRUE16-NEXT: v_maxmin_num_f16 v0.l, v0.l, v0.h, v1.l
+; GISEL-GFX1250-TRUE16-NEXT: v_maxmin_num_f16 v0.l, v0.l, v1.l, v2.l
; GISEL-GFX1250-TRUE16-NEXT: s_set_pc_i64 s[30:31]
;
; GISEL-GFX1250-FAKE16-LABEL: test_minmax_commuted_f16_ieee_true:
; GISEL-GFX1250-FAKE16: ; %bb.0:
; GISEL-GFX1250-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
; GISEL-GFX1250-FAKE16-NEXT: s_wait_kmcnt 0x0
-; GISEL-GFX1250-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GISEL-GFX1250-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GISEL-GFX1250-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
; GISEL-GFX1250-FAKE16-NEXT: v_maxmin_num_f16 v0, v0, v1, v2
; GISEL-GFX1250-FAKE16-NEXT: s_set_pc_i64 s[30:31]
%max = call half @llvm.maxnum.f16(half %a, half %b)
@@ -1524,18 +1448,12 @@ define half @test_maxmin_commuted_f16_ieee_true(half %a, half %b, half %c) {
; GISEL-GFX1170-TRUE16-LABEL: test_maxmin_commuted_f16_ieee_true:
; GISEL-GFX1170-TRUE16: ; %bb.0:
; GISEL-GFX1170-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GISEL-GFX1170-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GISEL-GFX1170-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v1.l, v1.l
-; GISEL-GFX1170-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v2.l, v2.l
-; GISEL-GFX1170-TRUE16-NEXT: v_minmax_num_f16 v0.l, v0.l, v0.h, v1.l
+; GISEL-GFX1170-TRUE16-NEXT: v_minmax_num_f16 v0.l, v0.l, v1.l, v2.l
; GISEL-GFX1170-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GISEL-GFX1170-FAKE16-LABEL: test_maxmin_commuted_f16_ieee_true:
; GISEL-GFX1170-FAKE16: ; %bb.0:
; GISEL-GFX1170-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GISEL-GFX1170-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GISEL-GFX1170-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GISEL-GFX1170-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
; GISEL-GFX1170-FAKE16-NEXT: v_minmax_num_f16 v0, v0, v1, v2
; GISEL-GFX1170-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1566,10 +1484,7 @@ define half @test_maxmin_commuted_f16_ieee_true(half %a, half %b, half %c) {
; GISEL-GFX12-TRUE16-NEXT: s_wait_samplecnt 0x0
; GISEL-GFX12-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GISEL-GFX12-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GISEL-GFX12-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GISEL-GFX12-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v1.l, v1.l
-; GISEL-GFX12-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v2.l, v2.l
-; GISEL-GFX12-TRUE16-NEXT: v_minmax_num_f16 v0.l, v0.l, v0.h, v1.l
+; GISEL-GFX12-TRUE16-NEXT: v_minmax_num_f16 v0.l, v0.l, v1.l, v2.l
; GISEL-GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GISEL-GFX12-FAKE16-LABEL: test_maxmin_commuted_f16_ieee_true:
@@ -1579,9 +1494,6 @@ define half @test_maxmin_commuted_f16_ieee_true(half %a, half %b, half %c) {
; GISEL-GFX12-FAKE16-NEXT: s_wait_samplecnt 0x0
; GISEL-GFX12-FAKE16-NEXT: s_wait_bvhcnt 0x0
; GISEL-GFX12-FAKE16-NEXT: s_wait_kmcnt 0x0
-; GISEL-GFX12-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GISEL-GFX12-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GISEL-GFX12-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
; GISEL-GFX12-FAKE16-NEXT: v_minmax_num_f16 v0, v0, v1, v2
; GISEL-GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1603,19 +1515,13 @@ define half @test_maxmin_commuted_f16_ieee_true(half %a, half %b, half %c) {
; GISEL-GFX1250-TRUE16: ; %bb.0:
; GISEL-GFX1250-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
; GISEL-GFX1250-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GISEL-GFX1250-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GISEL-GFX1250-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v1.l, v1.l
-; GISEL-GFX1250-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v2.l, v2.l
-; GISEL-GFX1250-TRUE16-NEXT: v_minmax_num_f16 v0.l, v0.l, v0.h, v1.l
+; GISEL-GFX1250-TRUE16-NEXT: v_minmax_num_f16 v0.l, v0.l, v1.l, v2.l
; GISEL-GFX1250-TRUE16-NEXT: s_set_pc_i64 s[30:31]
;
; GISEL-GFX1250-FAKE16-LABEL: test_maxmin_commuted_f16_ieee_true:
; GISEL-GFX1250-FAKE16: ; %bb.0:
; GISEL-GFX1250-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
; GISEL-GFX1250-FAKE16-NEXT: s_wait_kmcnt 0x0
-; GISEL-GFX1250-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GISEL-GFX1250-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GISEL-GFX1250-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
; GISEL-GFX1250-FAKE16-NEXT: v_minmax_num_f16 v0, v0, v1, v2
; GISEL-GFX1250-FAKE16-NEXT: s_set_pc_i64 s[30:31]
%min = call half @llvm.minnum.f16(half %a, half %b)
diff --git a/llvm/test/CodeGen/AMDGPU/packed-fneg-fsub-fp16.ll b/llvm/test/CodeGen/AMDGPU/packed-fneg-fsub-fp16.ll
index 9ff0714564fc3..0afdd729b0200 100644
--- a/llvm/test/CodeGen/AMDGPU/packed-fneg-fsub-fp16.ll
+++ b/llvm/test/CodeGen/AMDGPU/packed-fneg-fsub-fp16.ll
@@ -313,26 +313,13 @@ define <4 x half> @fminnum_v4f16_neg(<4 x half> %first, <4 x half> %second) {
; GFX950-GISEL-NEXT: v_pk_min_f16 v1, v1, v2
; GFX950-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX1250-SDAG-LABEL: fminnum_v4f16_neg:
-; GFX1250-SDAG: ; %bb.0:
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX1250-SDAG-NEXT: v_pk_min_num_f16 v0, v0, v2 neg_lo:[0,1] neg_hi:[0,1]
-; GFX1250-SDAG-NEXT: v_pk_min_num_f16 v1, v1, v3 neg_lo:[0,1] neg_hi:[0,1]
-; GFX1250-SDAG-NEXT: s_set_pc_i64 s[30:31]
-;
-; GFX1250-GISEL-LABEL: fminnum_v4f16_neg:
-; GFX1250-GISEL: ; %bb.0:
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v0, v0, v0
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v2, v2, v2 neg_lo:[1,1] neg_hi:[1,1]
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v1, v1, v1
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v3, v3, v3 neg_lo:[1,1] neg_hi:[1,1]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1250-GISEL-NEXT: v_pk_min_num_f16 v0, v0, v2
-; GFX1250-GISEL-NEXT: v_pk_min_num_f16 v1, v1, v3
-; GFX1250-GISEL-NEXT: s_set_pc_i64 s[30:31]
+; GFX1250-LABEL: fminnum_v4f16_neg:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: v_pk_min_num_f16 v0, v0, v2 neg_lo:[0,1] neg_hi:[0,1]
+; GFX1250-NEXT: v_pk_min_num_f16 v1, v1, v3 neg_lo:[0,1] neg_hi:[0,1]
+; GFX1250-NEXT: s_set_pc_i64 s[30:31]
%neg = fneg <4 x half> %second
%fmin = tail call <4 x half> @llvm.minnum.v4f16(<4 x half> %first, <4 x half> %neg)
ret <4 x half> %fmin
@@ -375,34 +362,15 @@ define <8 x half> @fminnum_v8f16_neg(<8 x half> %first, <8 x half> %second) {
; GFX950-GISEL-NEXT: v_pk_min_f16 v3, v3, v4
; GFX950-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX1250-SDAG-LABEL: fminnum_v8f16_neg:
-; GFX1250-SDAG: ; %bb.0:
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX1250-SDAG-NEXT: v_pk_min_num_f16 v0, v0, v4 neg_lo:[0,1] neg_hi:[0,1]
-; GFX1250-SDAG-NEXT: v_pk_min_num_f16 v1, v1, v5 neg_lo:[0,1] neg_hi:[0,1]
-; GFX1250-SDAG-NEXT: v_pk_min_num_f16 v2, v2, v6 neg_lo:[0,1] neg_hi:[0,1]
-; GFX1250-SDAG-NEXT: v_pk_min_num_f16 v3, v3, v7 neg_lo:[0,1] neg_hi:[0,1]
-; GFX1250-SDAG-NEXT: s_set_pc_i64 s[30:31]
-;
-; GFX1250-GISEL-LABEL: fminnum_v8f16_neg:
-; GFX1250-GISEL: ; %bb.0:
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v0, v0, v0
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v4, v4, v4 neg_lo:[1,1] neg_hi:[1,1]
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v1, v1, v1
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v5, v5, v5 neg_lo:[1,1] neg_hi:[1,1]
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v2, v2, v2
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v6, v6, v6 neg_lo:[1,1] neg_hi:[1,1]
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v3, v3, v3
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v7, v7, v7 neg_lo:[1,1] neg_hi:[1,1]
-; GFX1250-GISEL-NEXT: v_pk_min_num_f16 v0, v0, v4
-; GFX1250-GISEL-NEXT: v_pk_min_num_f16 v1, v1, v5
-; GFX1250-GISEL-NEXT: v_pk_min_num_f16 v2, v2, v6
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4)
-; GFX1250-GISEL-NEXT: v_pk_min_num_f16 v3, v3, v7
-; GFX1250-GISEL-NEXT: s_set_pc_i64 s[30:31]
+; GFX1250-LABEL: fminnum_v8f16_neg:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: v_pk_min_num_f16 v0, v0, v4 neg_lo:[0,1] neg_hi:[0,1]
+; GFX1250-NEXT: v_pk_min_num_f16 v1, v1, v5 neg_lo:[0,1] neg_hi:[0,1]
+; GFX1250-NEXT: v_pk_min_num_f16 v2, v2, v6 neg_lo:[0,1] neg_hi:[0,1]
+; GFX1250-NEXT: v_pk_min_num_f16 v3, v3, v7 neg_lo:[0,1] neg_hi:[0,1]
+; GFX1250-NEXT: s_set_pc_i64 s[30:31]
%neg = fneg <8 x half> %second
%fmin = tail call <8 x half> @llvm.minnum.v8f16(<8 x half> %first, <8 x half> %neg)
ret <8 x half> %fmin
@@ -433,26 +401,13 @@ define <4 x half> @fmaxnum_v4f16_neg(<4 x half> %first, <4 x half> %second) {
; GFX950-GISEL-NEXT: v_pk_max_f16 v1, v1, v2
; GFX950-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX1250-SDAG-LABEL: fmaxnum_v4f16_neg:
-; GFX1250-SDAG: ; %bb.0:
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX1250-SDAG-NEXT: v_pk_max_num_f16 v0, v0, v2 neg_lo:[0,1] neg_hi:[0,1]
-; GFX1250-SDAG-NEXT: v_pk_max_num_f16 v1, v1, v3 neg_lo:[0,1] neg_hi:[0,1]
-; GFX1250-SDAG-NEXT: s_set_pc_i64 s[30:31]
-;
-; GFX1250-GISEL-LABEL: fmaxnum_v4f16_neg:
-; GFX1250-GISEL: ; %bb.0:
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v0, v0, v0
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v2, v2, v2 neg_lo:[1,1] neg_hi:[1,1]
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v1, v1, v1
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v3, v3, v3 neg_lo:[1,1] neg_hi:[1,1]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v0, v0, v2
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v1, v1, v3
-; GFX1250-GISEL-NEXT: s_set_pc_i64 s[30:31]
+; GFX1250-LABEL: fmaxnum_v4f16_neg:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: v_pk_max_num_f16 v0, v0, v2 neg_lo:[0,1] neg_hi:[0,1]
+; GFX1250-NEXT: v_pk_max_num_f16 v1, v1, v3 neg_lo:[0,1] neg_hi:[0,1]
+; GFX1250-NEXT: s_set_pc_i64 s[30:31]
%neg = fneg <4 x half> %second
%fmax = tail call <4 x half> @llvm.maxnum.v4f16(<4 x half> %first, <4 x half> %neg)
ret <4 x half> %fmax
@@ -495,35 +450,19 @@ define <8 x half> @fmaxnum_v8f16_neg(<8 x half> %first, <8 x half> %second) {
; GFX950-GISEL-NEXT: v_pk_max_f16 v3, v3, v4
; GFX950-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX1250-SDAG-LABEL: fmaxnum_v8f16_neg:
-; GFX1250-SDAG: ; %bb.0:
-; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX1250-SDAG-NEXT: v_pk_max_num_f16 v0, v0, v4 neg_lo:[0,1] neg_hi:[0,1]
-; GFX1250-SDAG-NEXT: v_pk_max_num_f16 v1, v1, v5 neg_lo:[0,1] neg_hi:[0,1]
-; GFX1250-SDAG-NEXT: v_pk_max_num_f16 v2, v2, v6 neg_lo:[0,1] neg_hi:[0,1]
-; GFX1250-SDAG-NEXT: v_pk_max_num_f16 v3, v3, v7 neg_lo:[0,1] neg_hi:[0,1]
-; GFX1250-SDAG-NEXT: s_set_pc_i64 s[30:31]
-;
-; GFX1250-GISEL-LABEL: fmaxnum_v8f16_neg:
-; GFX1250-GISEL: ; %bb.0:
-; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v0, v0, v0
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v4, v4, v4 neg_lo:[1,1] neg_hi:[1,1]
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v1, v1, v1
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v5, v5, v5 neg_lo:[1,1] neg_hi:[1,1]
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v2, v2, v2
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v6, v6, v6 neg_lo:[1,1] neg_hi:[1,1]
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v3, v3, v3
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v7, v7, v7 neg_lo:[1,1] neg_hi:[1,1]
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v0, v0, v4
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v1, v1, v5
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v2, v2, v6
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4)
-; GFX1250-GISEL-NEXT: v_pk_max_num_f16 v3, v3, v7
-; GFX1250-GISEL-NEXT: s_set_pc_i64 s[30:31]
+; GFX1250-LABEL: fmaxnum_v8f16_neg:
+; GFX1250: ; %bb.0:
+; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT: s_wait_kmcnt 0x0
+; GFX1250-NEXT: v_pk_max_num_f16 v0, v0, v4 neg_lo:[0,1] neg_hi:[0,1]
+; GFX1250-NEXT: v_pk_max_num_f16 v1, v1, v5 neg_lo:[0,1] neg_hi:[0,1]
+; GFX1250-NEXT: v_pk_max_num_f16 v2, v2, v6 neg_lo:[0,1] neg_hi:[0,1]
+; GFX1250-NEXT: v_pk_max_num_f16 v3, v3, v7 neg_lo:[0,1] neg_hi:[0,1]
+; GFX1250-NEXT: s_set_pc_i64 s[30:31]
%neg = fneg <8 x half> %second
%fmax = tail call <8 x half> @llvm.maxnum.v8f16(<8 x half> %first, <8 x half> %neg)
ret <8 x half> %fmax
}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX1250-GISEL: {{.*}}
+; GFX1250-SDAG: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/packed-fp64-uniform-vgpr-splat-operand.ll b/llvm/test/CodeGen/AMDGPU/packed-fp64-uniform-vgpr-splat-operand.ll
index f5e5ff70fb6fa..7c766b12e0c49 100644
--- a/llvm/test/CodeGen/AMDGPU/packed-fp64-uniform-vgpr-splat-operand.ll
+++ b/llvm/test/CodeGen/AMDGPU/packed-fp64-uniform-vgpr-splat-operand.ll
@@ -94,18 +94,16 @@ define <2 x double> @test_v2f64_fmul_fmul(<2 x double> %vec, double inreg %a) {
; GFX1251-GISEL: ; %bb.0: ; %entry
; GFX1251-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1251-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX1251-GISEL-NEXT: v_mul_f64_e64 v[8:9], 0x40240000, s[0:1]
+; GFX1251-GISEL-NEXT: v_mul_f64_e64 v[4:5], 0x40240000, s[0:1]
; GFX1251-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1251-GISEL-NEXT: v_readfirstlane_b32 s0, v8
-; GFX1251-GISEL-NEXT: v_readfirstlane_b32 s1, v9
+; GFX1251-GISEL-NEXT: v_readfirstlane_b32 s0, v4
+; GFX1251-GISEL-NEXT: v_readfirstlane_b32 s1, v5
; GFX1251-GISEL-NEXT: s_mov_b64 s[2:3], s[0:1]
; GFX1251-GISEL-NEXT: v_mov_b64_e32 v[4:5], s[0:1]
; GFX1251-GISEL-NEXT: v_mov_b64_e32 v[6:7], s[2:3]
; GFX1251-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1251-GISEL-NEXT: v_pk_add_f64 v[0:3], v[0:3], v[4:7]
-; GFX1251-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[8:9]
-; GFX1251-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
-; GFX1251-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[8:9]
+; GFX1251-GISEL-NEXT: v_pk_max_num_f64 v[0:3], v[0:3], v[4:7]
; GFX1251-GISEL-NEXT: s_set_pc_i64 s[30:31]
entry:
; mul0 is uniform but in vgpr
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-fmax.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-fmax.ll
index 9723357a39562..e727cae430b75 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-fmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-fmax.ll
@@ -147,9 +147,6 @@ define half @test_vector_reduce_fmax_v2half(<2 x half> %v) {
; GFX1170-GISEL-TRUE16-LABEL: test_vector_reduce_fmax_v2half:
; GFX1170-GISEL-TRUE16: ; %bb.0: ; %entry
; GFX1170-GISEL-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
-; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.h
; GFX1170-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -157,9 +154,7 @@ define half @test_vector_reduce_fmax_v2half(<2 x half> %v) {
; GFX1170-GISEL-FAKE16: ; %bb.0: ; %entry
; GFX1170-GISEL-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 16, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
+; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v1
; GFX1170-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -194,9 +189,6 @@ define half @test_vector_reduce_fmax_v2half(<2 x half> %v) {
; GFX12-GISEL-TRUE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.h
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -208,9 +200,7 @@ define half @test_vector_reduce_fmax_v2half(<2 x half> %v) {
; GFX12-GISEL-FAKE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 16, v0
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v1
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -367,10 +357,6 @@ define half @test_vector_reduce_fmax_v3half(<3 x half> %v) {
; GFX1170-GISEL-TRUE16-LABEL: test_vector_reduce_fmax_v3half:
; GFX1170-GISEL-TRUE16: ; %bb.0: ; %entry
; GFX1170-GISEL-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX1170-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -378,10 +364,7 @@ define half @test_vector_reduce_fmax_v3half(<3 x half> %v) {
; GFX1170-GISEL-FAKE16: ; %bb.0: ; %entry
; GFX1170-GISEL-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v2, 16, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
+; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v0, v0, v2, v1
; GFX1170-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -424,10 +407,6 @@ define half @test_vector_reduce_fmax_v3half(<3 x half> %v) {
; GFX12-GISEL-TRUE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -439,10 +418,7 @@ define half @test_vector_reduce_fmax_v3half(<3 x half> %v) {
; GFX12-GISEL-FAKE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v2, 16, v0
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v0, v0, v2, v1
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -627,12 +603,8 @@ define half @test_vector_reduce_fmax_v4half(<4 x half> %v) {
; GFX1170-GISEL-TRUE16-LABEL: test_vector_reduce_fmax_v4half:
; GFX1170-GISEL-TRUE16: ; %bb.0: ; %entry
; GFX1170-GISEL-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v1.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
-; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.h
+; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX1170-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -641,11 +613,6 @@ define half @test_vector_reduce_fmax_v4half(<4 x half> %v) {
; GFX1170-GISEL-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v3
; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v2
; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v0, v0, v3, v1
@@ -684,12 +651,8 @@ define half @test_vector_reduce_fmax_v4half(<4 x half> %v) {
; GFX12-GISEL-TRUE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v1.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.h
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -702,11 +665,6 @@ define half @test_vector_reduce_fmax_v4half(<4 x half> %v) {
; GFX12-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v3
; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v2
; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v0, v0, v3, v1
@@ -1000,19 +958,11 @@ define half @test_vector_reduce_fmax_v8half(<8 x half> %v) {
; GFX1170-GISEL-TRUE16-LABEL: test_vector_reduce_fmax_v8half:
; GFX1170-GISEL-TRUE16: ; %bb.0: ; %entry
; GFX1170-GISEL-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v1.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.l, v3.l, v3.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.h, v3.h, v3.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v2.l, v2.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.l, v2.h, v2.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.h, v3.l, v3.h
-; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v3.l, v3.h
+; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.l, v0.l, v0.h, v1.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.h, v1.h, v2.l, v2.h
+; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.h, v2.l, v2.h, v1.h
; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.h
; GFX1170-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -1020,23 +970,16 @@ define half @test_vector_reduce_fmax_v8half(<8 x half> %v) {
; GFX1170-GISEL-FAKE16-LABEL: test_vector_reduce_fmax_v8half:
; GFX1170-GISEL-FAKE16: ; %bb.0: ; %entry
; GFX1170-GISEL-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v5, 16, v1
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v7, 16, v3
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v4, 16, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v6, 16, v2
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v5, v5
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v3
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v7, v7, v7
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v4, v4, v4
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v5
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v6, v6
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v7
-; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v0, v0, v4, v1
-; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v1, v2, v5, v3
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v4, 16, v1
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v5, 16, v3
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v6, 16, v0
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v7, 16, v2
+; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v4
+; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v5
+; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v0, v0, v6, v1
+; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v1, v2, v7, v3
; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v1
; GFX1170-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -1080,19 +1023,11 @@ define half @test_vector_reduce_fmax_v8half(<8 x half> %v) {
; GFX12-GISEL-TRUE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v1.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.l, v3.l, v3.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.h, v3.h, v3.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v2.l, v2.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.l, v2.h, v2.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.h, v3.l, v3.h
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v3.l, v3.h
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.l, v0.l, v0.h, v1.l
-; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.h, v1.h, v2.l, v2.h
+; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.h, v2.l, v2.h, v1.h
; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.h
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -1104,23 +1039,16 @@ define half @test_vector_reduce_fmax_v8half(<8 x half> %v) {
; GFX12-GISEL-FAKE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-FAKE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v5, 16, v1
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v7, 16, v3
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v4, 16, v0
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v6, 16, v2
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v5, v5
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v3
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v7, v7, v7
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v4, v4, v4
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v5
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v6, v6
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v7
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v0, v0, v4, v1
-; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v1, v2, v5, v3
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v4, 16, v1
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v5, 16, v3
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v6, 16, v0
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v7, 16, v2
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v4
+; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v5
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v0, v0, v6, v1
+; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v1, v2, v7, v3
; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v1
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -1621,32 +1549,17 @@ define half @test_vector_reduce_fmax_v16half(<16 x half> %v) {
; GFX1170-GISEL-TRUE16-LABEL: test_vector_reduce_fmax_v16half:
; GFX1170-GISEL-TRUE16: ; %bb.0: ; %entry
; GFX1170-GISEL-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v1.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v4.l, v4.l, v4.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v4.h, v4.h, v4.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
+; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.l, v5.l, v5.h
+; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.h, v7.l, v7.h
; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v3.l, v3.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.l, v3.h, v3.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.h, v5.l, v5.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.l, v5.h, v5.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.h, v7.l, v7.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v7.l, v7.h, v7.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.l, v2.l, v2.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.h, v2.h, v2.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.h, v3.h, v5.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.l, v6.l, v6.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v6.l, v6.h, v6.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.h, v5.h, v7.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v3.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v3.l, v4.l, v4.h, v3.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.l, v0.l, v0.h, v1.l
+; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v3.l, v3.h
+; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v3.l, v4.l, v4.h, v5.l
+; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v3.h, v6.l, v6.h, v5.h
; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v3.h, v5.l, v6.l, v5.h
+; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.h, v2.l, v2.h, v1.h
-; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v3.l, v3.h
; GFX1170-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX1170-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -1654,41 +1567,25 @@ define half @test_vector_reduce_fmax_v16half(<16 x half> %v) {
; GFX1170-GISEL-FAKE16-LABEL: test_vector_reduce_fmax_v16half:
; GFX1170-GISEL-FAKE16: ; %bb.0: ; %entry
; GFX1170-GISEL-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v10, 16, v5
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v11, 16, v7
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v9, 16, v1
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v11, 16, v3
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v13, 16, v5
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v15, 16, v7
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v9, v9, v9
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v12, 16, v4
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v12, 16, v3
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v13, 16, v4
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v14, 16, v6
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v5, v5
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v7, v7, v7
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v9
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v9, v11, v11
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v11, v13, v13
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v13, v15, v15
+; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v5, v10
+; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v7, v7, v11
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v8, 16, v0
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v10, 16, v2
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v3
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v4, v4, v4
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v12, v12, v12
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v5, v11
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v6, v6, v6
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v11, v14, v14
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v7, v7, v13
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v8, v8, v8
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v10, v10, v10
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v9
-; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v4, v4, v12, v5
-; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v5, v6, v11, v7
+; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v9
+; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v12
+; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v4, v4, v13, v5
+; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v5, v6, v14, v7
+; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v0, v0, v8, v1
-; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v1, v2, v10, v3
+; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v4, v5
-; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-FAKE16-NEXT: v_max3_num_f16 v0, v0, v1, v2
; GFX1170-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1741,32 +1638,17 @@ define half @test_vector_reduce_fmax_v16half(<16 x half> %v) {
; GFX12-GISEL-TRUE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v1.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v4.l, v4.l, v4.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v4.h, v4.h, v4.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
+; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.l, v5.l, v5.h
+; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.h, v7.l, v7.h
; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v3.l, v3.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.l, v3.h, v3.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.h, v5.l, v5.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.l, v5.h, v5.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.h, v7.l, v7.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v7.l, v7.h, v7.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.l, v2.l, v2.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.h, v2.h, v2.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.h, v3.h, v5.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.l, v6.l, v6.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v6.l, v6.h, v6.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.h, v5.h, v7.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v3.l
-; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v3.l, v4.l, v4.h, v3.h
-; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.l, v0.l, v0.h, v1.l
+; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v3.l, v3.h
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v3.l, v4.l, v4.h, v5.l
+; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v3.h, v6.l, v6.h, v5.h
; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v3.h, v5.l, v6.l, v5.h
+; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.h, v2.l, v2.h, v1.h
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v3.l, v3.h
; GFX12-GISEL-TRUE16-NEXT: v_max3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -1778,41 +1660,25 @@ define half @test_vector_reduce_fmax_v16half(<16 x half> %v) {
; GFX12-GISEL-FAKE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-FAKE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v10, 16, v5
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v11, 16, v7
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v9, 16, v1
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v11, 16, v3
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v13, 16, v5
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v15, 16, v7
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v9, v9, v9
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v12, 16, v4
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v12, 16, v3
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v13, 16, v4
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v14, 16, v6
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v5, v5
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v7, v7, v7
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v9
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v9, v11, v11
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v11, v13, v13
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v13, v15, v15
+; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v5, v10
+; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v7, v7, v11
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v8, 16, v0
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v10, 16, v2
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v3
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v4, v4, v4
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v12, v12, v12
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v5, v11
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v6, v6, v6
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v11, v14, v14
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v7, v7, v13
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v8, v8, v8
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v10, v10, v10
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v9
-; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v4, v4, v12, v5
-; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v5, v6, v11, v7
+; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v9
+; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v12
+; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v4, v4, v13, v5
+; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v5, v6, v14, v7
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v0, v0, v8, v1
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v1, v2, v10, v3
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v4, v5
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_max3_num_f16 v0, v0, v1, v2
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -1901,41 +1767,21 @@ define float @test_vector_reduce_fmax_v2float(<2 x float> %v) {
; GFX11-GISEL-NEXT: v_max_f32_e32 v0, v0, v1
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX1170-SDAG-LABEL: test_vector_reduce_fmax_v2float:
-; GFX1170-SDAG: ; %bb.0: ; %entry
-; GFX1170-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-SDAG-NEXT: v_max_num_f32_e32 v0, v0, v1
-; GFX1170-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX1170-GISEL-LABEL: test_vector_reduce_fmax_v2float:
-; GFX1170-GISEL: ; %bb.0: ; %entry
-; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v0, v0, v1
-; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-SDAG-LABEL: test_vector_reduce_fmax_v2float:
-; GFX12-SDAG: ; %bb.0: ; %entry
-; GFX12-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-SDAG-NEXT: s_wait_expcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_samplecnt 0x0
-; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: v_max_num_f32_e32 v0, v0, v1
-; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-GISEL-LABEL: test_vector_reduce_fmax_v2float:
-; GFX12-GISEL: ; %bb.0: ; %entry
-; GFX12-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-GISEL-NEXT: s_wait_expcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
-; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_max_num_f32_e32 v0, v0, v1
-; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
+; GFX1170-LABEL: test_vector_reduce_fmax_v2float:
+; GFX1170: ; %bb.0: ; %entry
+; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1170-NEXT: v_max_num_f32_e32 v0, v0, v1
+; GFX1170-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX12-LABEL: test_vector_reduce_fmax_v2float:
+; GFX12: ; %bb.0: ; %entry
+; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT: s_wait_expcnt 0x0
+; GFX12-NEXT: s_wait_samplecnt 0x0
+; GFX12-NEXT: s_wait_bvhcnt 0x0
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: v_max_num_f32_e32 v0, v0, v1
+; GFX12-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call float @llvm.vector.reduce.fmax.v2float(<2 x float> %v)
ret float %res
@@ -2017,43 +1863,21 @@ define float @test_vector_reduce_fmax_v3float(<3 x float> %v) {
; GFX11-GISEL-NEXT: v_max3_f32 v0, v0, v1, v2
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX1170-SDAG-LABEL: test_vector_reduce_fmax_v3float:
-; GFX1170-SDAG: ; %bb.0: ; %entry
-; GFX1170-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-SDAG-NEXT: v_max3_num_f32 v0, v0, v1, v2
-; GFX1170-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX1170-GISEL-LABEL: test_vector_reduce_fmax_v3float:
-; GFX1170-GISEL: ; %bb.0: ; %entry
-; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v2, v2, v2
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1170-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
-; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-SDAG-LABEL: test_vector_reduce_fmax_v3float:
-; GFX12-SDAG: ; %bb.0: ; %entry
-; GFX12-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-SDAG-NEXT: s_wait_expcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_samplecnt 0x0
-; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: v_max3_num_f32 v0, v0, v1, v2
-; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-GISEL-LABEL: test_vector_reduce_fmax_v3float:
-; GFX12-GISEL: ; %bb.0: ; %entry
-; GFX12-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-GISEL-NEXT: s_wait_expcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
-; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-GISEL-NEXT: v_max_num_f32_e32 v2, v2, v2
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
-; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
+; GFX1170-LABEL: test_vector_reduce_fmax_v3float:
+; GFX1170: ; %bb.0: ; %entry
+; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1170-NEXT: v_max3_num_f32 v0, v0, v1, v2
+; GFX1170-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX12-LABEL: test_vector_reduce_fmax_v3float:
+; GFX12: ; %bb.0: ; %entry
+; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT: s_wait_expcnt 0x0
+; GFX12-NEXT: s_wait_samplecnt 0x0
+; GFX12-NEXT: s_wait_bvhcnt 0x0
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: v_max3_num_f32 v0, v0, v1, v2
+; GFX12-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call float @llvm.vector.reduce.fmax.v3float(<3 x float> %v)
ret float %res
@@ -2170,10 +1994,8 @@ define float @test_vector_reduce_fmax_v4float(<4 x float> %v) {
; GFX1170-GISEL-LABEL: test_vector_reduce_fmax_v4float:
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v2, v2, v3
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -2196,10 +2018,8 @@ define float @test_vector_reduce_fmax_v4float(<4 x float> %v) {
; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_max_num_f32_e32 v2, v2, v3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -2366,15 +2186,11 @@ define float @test_vector_reduce_fmax_v8float(<8 x float> %v) {
; GFX1170-GISEL-LABEL: test_vector_reduce_fmax_v8float:
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v7, v7, v7
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v6, v6, v6 :: v_dual_max_num_f32 v1, v1, v1
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v3 :: v_dual_max_num_f32 v3, v4, v4
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v4, v5, v5 :: v_dual_max_num_f32 v5, v6, v7
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v2, v2, v3
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
-; GFX1170-GISEL-NEXT: v_max3_num_f32 v1, v3, v4, v5
+; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v3, v6, v7
+; GFX1170-GISEL-NEXT: v_max3_num_f32 v1, v4, v5, v3
; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v0, v0, v1
; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -2401,15 +2217,11 @@ define float @test_vector_reduce_fmax_v8float(<8 x float> %v) {
; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v7, v7, v7
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v6, v6, v6 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v3 :: v_dual_max_num_f32 v3, v4, v4
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v4, v5, v5 :: v_dual_max_num_f32 v5, v6, v7
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-GISEL-NEXT: v_max_num_f32_e32 v2, v2, v3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
-; GFX12-GISEL-NEXT: v_max3_num_f32 v1, v3, v4, v5
+; GFX12-GISEL-NEXT: v_max_num_f32_e32 v3, v6, v7
+; GFX12-GISEL-NEXT: v_max3_num_f32 v1, v4, v5, v3
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_max_num_f32_e32 v0, v0, v1
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -2667,25 +2479,17 @@ define float @test_vector_reduce_fmax_v16float(<16 x float> %v) {
; GFX1170-GISEL-LABEL: test_vector_reduce_fmax_v16float:
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v8, v8, v8
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_4)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v9, v9, v9 :: v_dual_max_num_f32 v2, v2, v3
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v3, v6, v6 :: v_dual_max_num_f32 v6, v7, v7
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v7, v10, v10 :: v_dual_max_num_f32 v10, v11, v11
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v11, v14, v14 :: v_dual_max_num_f32 v14, v15, v15
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v5, v5, v5 :: v_dual_max_num_f32 v4, v4, v4
-; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v3, v3, v6
+; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v10, v10, v11
+; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v11, v14, v15
+; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v2, v2, v3
+; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v3, v6, v7
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1170-GISEL-NEXT: v_max3_num_f32 v6, v8, v9, v10
+; GFX1170-GISEL-NEXT: v_max3_num_f32 v7, v12, v13, v11
; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v7, v7, v10 :: v_dual_max_num_f32 v10, v12, v12
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v12, v13, v13 :: v_dual_max_num_f32 v11, v11, v14
; GFX1170-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
-; GFX1170-GISEL-NEXT: v_max3_num_f32 v6, v8, v9, v7
; GFX1170-GISEL-NEXT: v_max3_num_f32 v1, v4, v5, v3
-; GFX1170-GISEL-NEXT: v_max3_num_f32 v7, v10, v12, v11
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v2, v6, v7
; GFX1170-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -2718,26 +2522,18 @@ define float @test_vector_reduce_fmax_v16float(<16 x float> %v) {
; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX12-GISEL-NEXT: v_max_num_f32_e32 v8, v8, v8
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_4)
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v9, v9, v9 :: v_dual_max_num_f32 v2, v2, v3
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v3, v6, v6 :: v_dual_max_num_f32 v6, v7, v7
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v7, v10, v10 :: v_dual_max_num_f32 v10, v11, v11
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v11, v14, v14 :: v_dual_max_num_f32 v14, v15, v15
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v5, v5, v5 :: v_dual_max_num_f32 v4, v4, v4
-; GFX12-GISEL-NEXT: v_max_num_f32_e32 v3, v3, v6
+; GFX12-GISEL-NEXT: v_max_num_f32_e32 v10, v10, v11
+; GFX12-GISEL-NEXT: v_max_num_f32_e32 v11, v14, v15
+; GFX12-GISEL-NEXT: v_max_num_f32_e32 v2, v2, v3
+; GFX12-GISEL-NEXT: v_max_num_f32_e32 v3, v6, v7
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v7, v7, v10 :: v_dual_max_num_f32 v10, v12, v12
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v12, v13, v13 :: v_dual_max_num_f32 v11, v11, v14
+; GFX12-GISEL-NEXT: v_max3_num_f32 v6, v8, v9, v10
+; GFX12-GISEL-NEXT: v_max3_num_f32 v7, v12, v13, v11
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
-; GFX12-GISEL-NEXT: v_max3_num_f32 v6, v8, v9, v7
-; GFX12-GISEL-NEXT: v_max3_num_f32 v1, v4, v5, v3
-; GFX12-GISEL-NEXT: v_max3_num_f32 v7, v10, v12, v11
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_max_num_f32_e32 v2, v6, v7
+; GFX12-GISEL-NEXT: v_max3_num_f32 v1, v4, v5, v3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -2829,43 +2625,21 @@ define double @test_vector_reduce_fmax_v2double(<2 x double> %v) {
; GFX11-GISEL-NEXT: v_max_f64 v[0:1], v[0:1], v[2:3]
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX1170-SDAG-LABEL: test_vector_reduce_fmax_v2double:
-; GFX1170-SDAG: ; %bb.0: ; %entry
-; GFX1170-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-SDAG-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
-; GFX1170-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX1170-GISEL-LABEL: test_vector_reduce_fmax_v2double:
-; GFX1170-GISEL: ; %bb.0: ; %entry
-; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[0:1]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[2:3], v[2:3]
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
-; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-SDAG-LABEL: test_vector_reduce_fmax_v2double:
-; GFX12-SDAG: ; %bb.0: ; %entry
-; GFX12-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-SDAG-NEXT: s_wait_expcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_samplecnt 0x0
-; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
-; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-GISEL-LABEL: test_vector_reduce_fmax_v2double:
-; GFX12-GISEL: ; %bb.0: ; %entry
-; GFX12-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-GISEL-NEXT: s_wait_expcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
-; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
-; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
+; GFX1170-LABEL: test_vector_reduce_fmax_v2double:
+; GFX1170: ; %bb.0: ; %entry
+; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1170-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
+; GFX1170-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX12-LABEL: test_vector_reduce_fmax_v2double:
+; GFX12: ; %bb.0: ; %entry
+; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT: s_wait_expcnt 0x0
+; GFX12-NEXT: s_wait_samplecnt 0x0
+; GFX12-NEXT: s_wait_bvhcnt 0x0
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
+; GFX12-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call double @llvm.vector.reduce.fmax.v2double(<2 x double> %v)
ret double %res
@@ -2974,51 +2748,25 @@ define double @test_vector_reduce_fmax_v3double(<3 x double> %v) {
; GFX11-GISEL-NEXT: v_max_f64 v[0:1], v[0:1], v[4:5]
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX1170-SDAG-LABEL: test_vector_reduce_fmax_v3double:
-; GFX1170-SDAG: ; %bb.0: ; %entry
-; GFX1170-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-SDAG-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
-; GFX1170-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1170-SDAG-NEXT: v_max_num_f64 v[0:1], v[0:1], v[4:5]
-; GFX1170-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX1170-GISEL-LABEL: test_vector_reduce_fmax_v3double:
-; GFX1170-GISEL: ; %bb.0: ; %entry
-; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[0:1]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[2:3], v[2:3]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[4:5], v[4:5], v[4:5]
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[4:5]
-; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-SDAG-LABEL: test_vector_reduce_fmax_v3double:
-; GFX12-SDAG: ; %bb.0: ; %entry
-; GFX12-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-SDAG-NEXT: s_wait_expcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_samplecnt 0x0
-; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-SDAG-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[4:5]
-; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-GISEL-LABEL: test_vector_reduce_fmax_v3double:
-; GFX12-GISEL: ; %bb.0: ; %entry
-; GFX12-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-GISEL-NEXT: s_wait_expcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
-; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[4:5]
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[4:5]
-; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
+; GFX1170-LABEL: test_vector_reduce_fmax_v3double:
+; GFX1170: ; %bb.0: ; %entry
+; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1170-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
+; GFX1170-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1170-NEXT: v_max_num_f64 v[0:1], v[0:1], v[4:5]
+; GFX1170-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX12-LABEL: test_vector_reduce_fmax_v3double:
+; GFX12: ; %bb.0: ; %entry
+; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT: s_wait_expcnt 0x0
+; GFX12-NEXT: s_wait_samplecnt 0x0
+; GFX12-NEXT: s_wait_bvhcnt 0x0
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[4:5]
+; GFX12-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call double @llvm.vector.reduce.fmax.v3double(<3 x double> %v)
ret double %res
@@ -3161,11 +2909,6 @@ define double @test_vector_reduce_fmax_v4double(<4 x double> %v) {
; GFX1170-GISEL-LABEL: test_vector_reduce_fmax_v4double:
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[0:1]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[2:3], v[2:3]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[4:5], v[4:5], v[4:5]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[6:7], v[6:7], v[6:7]
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[4:5], v[6:7]
; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -3192,11 +2935,6 @@ define double @test_vector_reduce_fmax_v4double(<4 x double> %v) {
; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[4:5]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[6:7], v[6:7], v[6:7]
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[4:5], v[6:7]
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -3432,22 +3170,14 @@ define double @test_vector_reduce_fmax_v8double(<8 x double> %v) {
; GFX1170-GISEL-LABEL: test_vector_reduce_fmax_v8double:
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[0:1]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[2:3], v[2:3]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[4:5], v[4:5], v[4:5]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[6:7], v[6:7], v[6:7]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[8:9], v[8:9], v[8:9]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[10:11], v[10:11], v[10:11]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[12:13], v[12:13], v[12:13]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[14:15], v[14:15], v[14:15]
; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[4:5], v[6:7]
; GFX1170-GISEL-NEXT: v_max_num_f64 v[4:5], v[8:9], v[10:11]
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1170-GISEL-NEXT: v_max_num_f64 v[6:7], v[12:13], v[14:15]
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[4:5], v[6:7]
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -3477,22 +3207,14 @@ define double @test_vector_reduce_fmax_v8double(<8 x double> %v) {
; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[4:5]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[6:7], v[6:7], v[6:7]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[8:9], v[8:9], v[8:9]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[10:11], v[10:11], v[10:11]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[12:13], v[12:13], v[12:13]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[14:15], v[14:15], v[14:15]
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[4:5], v[6:7]
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[8:9], v[10:11]
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[6:7], v[12:13], v[14:15]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[4:5], v[6:7]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -3924,21 +3646,6 @@ define double @test_vector_reduce_fmax_v16double(<16 x double> %v) {
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1170-GISEL-NEXT: scratch_load_b32 v31, off, s32
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[0:1]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[2:3], v[2:3]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[4:5], v[4:5], v[4:5]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[6:7], v[6:7], v[6:7]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[8:9], v[8:9], v[8:9]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[10:11], v[10:11], v[10:11]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[12:13], v[12:13], v[12:13]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[14:15], v[14:15], v[14:15]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[16:17], v[16:17], v[16:17]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[18:19], v[18:19], v[18:19]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[20:21], v[20:21], v[20:21]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[22:23], v[22:23], v[22:23]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[24:25], v[24:25], v[24:25]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[26:27], v[26:27], v[26:27]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[28:29], v[28:29], v[28:29]
; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[4:5], v[6:7]
; GFX1170-GISEL-NEXT: v_max_num_f64 v[4:5], v[8:9], v[10:11]
@@ -3952,12 +3659,11 @@ define double @test_vector_reduce_fmax_v16double(<16 x double> %v) {
; GFX1170-GISEL-NEXT: v_max_num_f64 v[4:5], v[8:9], v[10:11]
; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0)
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[30:31], v[30:31], v[30:31]
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_max_num_f64 v[14:15], v[28:29], v[30:31]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[6:7], v[12:13], v[14:15]
; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1170-GISEL-NEXT: v_max_num_f64 v[6:7], v[12:13], v[14:15]
; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[4:5], v[6:7]
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[2:3]
; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -4002,21 +3708,6 @@ define double @test_vector_reduce_fmax_v16double(<16 x double> %v) {
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-NEXT: scratch_load_b32 v31, off, s32
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[4:5]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[6:7], v[6:7], v[6:7]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[8:9], v[8:9], v[8:9]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[10:11], v[10:11], v[10:11]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[12:13], v[12:13], v[12:13]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[14:15], v[14:15], v[14:15]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[16:17], v[16:17], v[16:17]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[18:19], v[18:19], v[18:19]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[20:21], v[20:21], v[20:21]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[22:23], v[22:23], v[22:23]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[24:25], v[24:25], v[24:25]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[26:27], v[26:27], v[26:27]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[28:29], v[28:29], v[28:29]
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[4:5], v[6:7]
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[8:9], v[10:11]
@@ -4030,12 +3721,11 @@ define double @test_vector_reduce_fmax_v16double(<16 x double> %v) {
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[8:9], v[10:11]
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_loadcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[30:31], v[30:31], v[30:31]
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[14:15], v[28:29], v[30:31]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[6:7], v[12:13], v[14:15]
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[6:7], v[12:13], v[14:15]
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[4:5], v[6:7]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -4061,7 +3751,5 @@ declare double @llvm.vector.reduce.fmax.v16double(<16 x double>)
;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
; GFX10: {{.*}}
; GFX11: {{.*}}
-; GFX1170: {{.*}}
-; GFX12: {{.*}}
; GFX8: {{.*}}
; GFX9: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-fmin.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-fmin.ll
index 61e0c52c62e3e..ba7acc62331cf 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-fmin.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-fmin.ll
@@ -147,9 +147,6 @@ define half @test_vector_reduce_fmin_v2half(<2 x half> %v) {
; GFX1170-GISEL-TRUE16-LABEL: test_vector_reduce_fmin_v2half:
; GFX1170-GISEL-TRUE16: ; %bb.0: ; %entry
; GFX1170-GISEL-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
-; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v0.l, v0.l, v0.h
; GFX1170-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -157,9 +154,7 @@ define half @test_vector_reduce_fmin_v2half(<2 x half> %v) {
; GFX1170-GISEL-FAKE16: ; %bb.0: ; %entry
; GFX1170-GISEL-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 16, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
+; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v0, v0, v1
; GFX1170-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -194,9 +189,6 @@ define half @test_vector_reduce_fmin_v2half(<2 x half> %v) {
; GFX12-GISEL-TRUE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v0.l, v0.l, v0.h
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -208,9 +200,7 @@ define half @test_vector_reduce_fmin_v2half(<2 x half> %v) {
; GFX12-GISEL-FAKE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 16, v0
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v0, v0, v1
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -367,10 +357,6 @@ define half @test_vector_reduce_fmin_v3half(<3 x half> %v) {
; GFX1170-GISEL-TRUE16-LABEL: test_vector_reduce_fmin_v3half:
; GFX1170-GISEL-TRUE16: ; %bb.0: ; %entry
; GFX1170-GISEL-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX1170-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -378,10 +364,7 @@ define half @test_vector_reduce_fmin_v3half(<3 x half> %v) {
; GFX1170-GISEL-FAKE16: ; %bb.0: ; %entry
; GFX1170-GISEL-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v2, 16, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
+; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v0, v0, v2, v1
; GFX1170-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -424,10 +407,6 @@ define half @test_vector_reduce_fmin_v3half(<3 x half> %v) {
; GFX12-GISEL-TRUE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -439,10 +418,7 @@ define half @test_vector_reduce_fmin_v3half(<3 x half> %v) {
; GFX12-GISEL-FAKE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v2, 16, v0
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v0, v0, v2, v1
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -627,12 +603,8 @@ define half @test_vector_reduce_fmin_v4half(<4 x half> %v) {
; GFX1170-GISEL-TRUE16-LABEL: test_vector_reduce_fmin_v4half:
; GFX1170-GISEL-TRUE16: ; %bb.0: ; %entry
; GFX1170-GISEL-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v1.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
-; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.l, v1.l, v1.h
+; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX1170-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -641,11 +613,6 @@ define half @test_vector_reduce_fmin_v4half(<4 x half> %v) {
; GFX1170-GISEL-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v3
; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v1, v1, v2
; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v0, v0, v3, v1
@@ -684,12 +651,8 @@ define half @test_vector_reduce_fmin_v4half(<4 x half> %v) {
; GFX12-GISEL-TRUE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v1.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.l, v1.l, v1.h
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -702,11 +665,6 @@ define half @test_vector_reduce_fmin_v4half(<4 x half> %v) {
; GFX12-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v3
; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v1, v1, v2
; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v0, v0, v3, v1
@@ -1000,19 +958,11 @@ define half @test_vector_reduce_fmin_v8half(<8 x half> %v) {
; GFX1170-GISEL-TRUE16-LABEL: test_vector_reduce_fmin_v8half:
; GFX1170-GISEL-TRUE16: ; %bb.0: ; %entry
; GFX1170-GISEL-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v1.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.l, v3.l, v3.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.h, v3.h, v3.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.l, v1.l, v1.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v2.l, v2.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.l, v2.h, v2.h
-; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v2.h, v3.l, v3.h
-; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.h, v3.l, v3.h
+; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.l, v0.l, v0.h, v1.l
-; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.h, v1.h, v2.l, v2.h
+; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.h, v2.l, v2.h, v1.h
; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v0.l, v0.l, v0.h
; GFX1170-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -1020,23 +970,16 @@ define half @test_vector_reduce_fmin_v8half(<8 x half> %v) {
; GFX1170-GISEL-FAKE16-LABEL: test_vector_reduce_fmin_v8half:
; GFX1170-GISEL-FAKE16: ; %bb.0: ; %entry
; GFX1170-GISEL-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v5, 16, v1
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v7, 16, v3
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v4, 16, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v6, 16, v2
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v5, v5
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v3
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v7, v7, v7
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v4, v4, v4
-; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v1, v1, v5
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v6, v6
-; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v3, v3, v7
-; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v0, v0, v4, v1
-; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v1, v2, v5, v3
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v4, 16, v1
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v5, 16, v3
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v6, 16, v0
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v7, 16, v2
+; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v1, v1, v4
+; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v3, v3, v5
+; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v0, v0, v6, v1
+; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v1, v2, v7, v3
; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v0, v0, v1
; GFX1170-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -1080,19 +1023,11 @@ define half @test_vector_reduce_fmin_v8half(<8 x half> %v) {
; GFX12-GISEL-TRUE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v1.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.l, v3.l, v3.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.h, v3.h, v3.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.l, v1.l, v1.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v2.l, v2.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.l, v2.h, v2.h
-; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v2.h, v3.l, v3.h
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.h, v3.l, v3.h
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.l, v0.l, v0.h, v1.l
-; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.h, v1.h, v2.l, v2.h
+; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.h, v2.l, v2.h, v1.h
; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v0.l, v0.l, v0.h
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -1104,23 +1039,16 @@ define half @test_vector_reduce_fmin_v8half(<8 x half> %v) {
; GFX12-GISEL-FAKE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-FAKE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v5, 16, v1
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v7, 16, v3
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v4, 16, v0
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v6, 16, v2
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v5, v5
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v3
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v7, v7, v7
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v4, v4, v4
-; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v1, v1, v5
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v6, v6
-; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v3, v3, v7
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v0, v0, v4, v1
-; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v1, v2, v5, v3
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v4, 16, v1
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v5, 16, v3
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v6, 16, v0
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v7, 16, v2
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v1, v1, v4
+; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v3, v3, v5
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v0, v0, v6, v1
+; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v1, v2, v7, v3
; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v0, v0, v1
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -1621,32 +1549,17 @@ define half @test_vector_reduce_fmin_v16half(<16 x half> %v) {
; GFX1170-GISEL-TRUE16-LABEL: test_vector_reduce_fmin_v16half:
; GFX1170-GISEL-TRUE16: ; %bb.0: ; %entry
; GFX1170-GISEL-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v1.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v4.l, v4.l, v4.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v4.h, v4.h, v4.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
+; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v5.l, v5.l, v5.h
+; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v5.h, v7.l, v7.h
; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.l, v1.l, v1.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v3.l, v3.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.l, v3.h, v3.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.h, v5.l, v5.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.l, v5.h, v5.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.h, v7.l, v7.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v7.l, v7.h, v7.h
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.l, v2.l, v2.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.h, v2.h, v2.h
-; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v3.h, v3.h, v5.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.l, v6.l, v6.l
-; GFX1170-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v6.l, v6.h, v6.h
-; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v5.h, v5.h, v7.l
-; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.h, v1.h, v3.l
-; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v3.l, v4.l, v4.h, v3.h
-; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.l, v0.l, v0.h, v1.l
+; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.h, v3.l, v3.h
+; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v3.l, v4.l, v4.h, v5.l
+; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v3.h, v6.l, v6.h, v5.h
; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v3.h, v5.l, v6.l, v5.h
+; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.h, v2.l, v2.h, v1.h
-; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1170-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.l, v3.l, v3.h
; GFX1170-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX1170-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -1654,41 +1567,25 @@ define half @test_vector_reduce_fmin_v16half(<16 x half> %v) {
; GFX1170-GISEL-FAKE16-LABEL: test_vector_reduce_fmin_v16half:
; GFX1170-GISEL-FAKE16: ; %bb.0: ; %entry
; GFX1170-GISEL-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v10, 16, v5
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v11, 16, v7
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v9, 16, v1
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v11, 16, v3
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v13, 16, v5
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v15, 16, v7
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v9, v9, v9
-; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v12, 16, v4
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v12, 16, v3
+; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v13, 16, v4
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v14, 16, v6
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v5, v5
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v7, v7, v7
-; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v1, v1, v9
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v9, v11, v11
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v11, v13, v13
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v13, v15, v15
+; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v5, v5, v10
+; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v7, v7, v11
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v8, 16, v0
; GFX1170-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v10, 16, v2
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v3
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v4, v4, v4
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v12, v12, v12
-; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v5, v5, v11
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v6, v6, v6
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v11, v14, v14
-; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v7, v7, v13
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v8, v8, v8
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
-; GFX1170-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v10, v10, v10
-; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v3, v3, v9
-; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v4, v4, v12, v5
-; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v5, v6, v11, v7
+; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v1, v1, v9
+; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v3, v3, v12
+; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v4, v4, v13, v5
+; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v5, v6, v14, v7
+; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v0, v0, v8, v1
-; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v1, v2, v10, v3
+; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v2, v4, v5
-; GFX1170-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-FAKE16-NEXT: v_min3_num_f16 v0, v0, v1, v2
; GFX1170-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1741,32 +1638,17 @@ define half @test_vector_reduce_fmin_v16half(<16 x half> %v) {
; GFX12-GISEL-TRUE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.l, v1.l, v1.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v1.h, v1.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v4.l, v4.l, v4.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v4.h, v4.h, v4.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.l, v0.l, v0.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v0.h, v0.h, v0.h
+; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v5.l, v5.l, v5.h
+; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v5.h, v7.l, v7.h
; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.l, v1.l, v1.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v1.h, v3.l, v3.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.l, v3.h, v3.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v3.h, v5.l, v5.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.l, v5.h, v5.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.h, v7.l, v7.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v7.l, v7.h, v7.h
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.l, v2.l, v2.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v2.h, v2.h, v2.h
-; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v3.h, v3.h, v5.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v5.l, v6.l, v6.l
-; GFX12-GISEL-TRUE16-NEXT: v_max_num_f16_e32 v6.l, v6.h, v6.h
-; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v5.h, v5.h, v7.l
-; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.h, v1.h, v3.l
-; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v3.l, v4.l, v4.h, v3.h
-; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.l, v0.l, v0.h, v1.l
+; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.h, v3.l, v3.h
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v3.l, v4.l, v4.h, v5.l
+; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v3.h, v6.l, v6.h, v5.h
; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v3.h, v5.l, v6.l, v5.h
+; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.h, v2.l, v2.h, v1.h
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_min_num_f16_e32 v1.l, v3.l, v3.h
; GFX12-GISEL-TRUE16-NEXT: v_min3_num_f16 v0.l, v0.l, v0.h, v1.l
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -1778,41 +1660,25 @@ define half @test_vector_reduce_fmin_v16half(<16 x half> %v) {
; GFX12-GISEL-FAKE16-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-FAKE16-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v10, 16, v5
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v11, 16, v7
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v9, 16, v1
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v11, 16, v3
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v13, 16, v5
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v1, v1, v1
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v15, 16, v7
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v9, v9, v9
-; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v12, 16, v4
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v12, 16, v3
+; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v13, 16, v4
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v14, 16, v6
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v5, v5, v5
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v7, v7, v7
-; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v1, v1, v9
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v9, v11, v11
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v11, v13, v13
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v13, v15, v15
+; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v5, v5, v10
+; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v7, v7, v11
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v8, 16, v0
; GFX12-GISEL-FAKE16-NEXT: v_lshrrev_b32_e32 v10, 16, v2
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v3, v3, v3
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v4, v4, v4
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v12, v12, v12
-; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v5, v5, v11
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v6, v6, v6
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v11, v14, v14
-; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v7, v7, v13
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v0, v0, v0
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v8, v8, v8
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v2, v2, v2
-; GFX12-GISEL-FAKE16-NEXT: v_max_num_f16_e32 v10, v10, v10
-; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v3, v3, v9
-; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v4, v4, v12, v5
-; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v5, v6, v11, v7
+; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v1, v1, v9
+; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v3, v3, v12
+; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v4, v4, v13, v5
+; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v5, v6, v14, v7
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v0, v0, v8, v1
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v1, v2, v10, v3
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_min_num_f16_e32 v2, v4, v5
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_min3_num_f16 v0, v0, v1, v2
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -1901,41 +1767,21 @@ define float @test_vector_reduce_fmin_v2float(<2 x float> %v) {
; GFX11-GISEL-NEXT: v_min_f32_e32 v0, v0, v1
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX1170-SDAG-LABEL: test_vector_reduce_fmin_v2float:
-; GFX1170-SDAG: ; %bb.0: ; %entry
-; GFX1170-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-SDAG-NEXT: v_min_num_f32_e32 v0, v0, v1
-; GFX1170-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX1170-GISEL-LABEL: test_vector_reduce_fmin_v2float:
-; GFX1170-GISEL: ; %bb.0: ; %entry
-; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v0, v0, v1
-; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-SDAG-LABEL: test_vector_reduce_fmin_v2float:
-; GFX12-SDAG: ; %bb.0: ; %entry
-; GFX12-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-SDAG-NEXT: s_wait_expcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_samplecnt 0x0
-; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: v_min_num_f32_e32 v0, v0, v1
-; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-GISEL-LABEL: test_vector_reduce_fmin_v2float:
-; GFX12-GISEL: ; %bb.0: ; %entry
-; GFX12-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-GISEL-NEXT: s_wait_expcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
-; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_min_num_f32_e32 v0, v0, v1
-; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
+; GFX1170-LABEL: test_vector_reduce_fmin_v2float:
+; GFX1170: ; %bb.0: ; %entry
+; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1170-NEXT: v_min_num_f32_e32 v0, v0, v1
+; GFX1170-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX12-LABEL: test_vector_reduce_fmin_v2float:
+; GFX12: ; %bb.0: ; %entry
+; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT: s_wait_expcnt 0x0
+; GFX12-NEXT: s_wait_samplecnt 0x0
+; GFX12-NEXT: s_wait_bvhcnt 0x0
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: v_min_num_f32_e32 v0, v0, v1
+; GFX12-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call float @llvm.vector.reduce.fmin.v2float(<2 x float> %v)
ret float %res
@@ -2017,43 +1863,21 @@ define float @test_vector_reduce_fmin_v3float(<3 x float> %v) {
; GFX11-GISEL-NEXT: v_min3_f32 v0, v0, v1, v2
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX1170-SDAG-LABEL: test_vector_reduce_fmin_v3float:
-; GFX1170-SDAG: ; %bb.0: ; %entry
-; GFX1170-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-SDAG-NEXT: v_min3_num_f32 v0, v0, v1, v2
-; GFX1170-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX1170-GISEL-LABEL: test_vector_reduce_fmin_v3float:
-; GFX1170-GISEL: ; %bb.0: ; %entry
-; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v2, v2, v2
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1170-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
-; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-SDAG-LABEL: test_vector_reduce_fmin_v3float:
-; GFX12-SDAG: ; %bb.0: ; %entry
-; GFX12-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-SDAG-NEXT: s_wait_expcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_samplecnt 0x0
-; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: v_min3_num_f32 v0, v0, v1, v2
-; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-GISEL-LABEL: test_vector_reduce_fmin_v3float:
-; GFX12-GISEL: ; %bb.0: ; %entry
-; GFX12-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-GISEL-NEXT: s_wait_expcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
-; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-GISEL-NEXT: v_max_num_f32_e32 v2, v2, v2
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
-; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
+; GFX1170-LABEL: test_vector_reduce_fmin_v3float:
+; GFX1170: ; %bb.0: ; %entry
+; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1170-NEXT: v_min3_num_f32 v0, v0, v1, v2
+; GFX1170-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX12-LABEL: test_vector_reduce_fmin_v3float:
+; GFX12: ; %bb.0: ; %entry
+; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT: s_wait_expcnt 0x0
+; GFX12-NEXT: s_wait_samplecnt 0x0
+; GFX12-NEXT: s_wait_bvhcnt 0x0
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: v_min3_num_f32 v0, v0, v1, v2
+; GFX12-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call float @llvm.vector.reduce.fmin.v3float(<3 x float> %v)
ret float %res
@@ -2170,10 +1994,8 @@ define float @test_vector_reduce_fmin_v4float(<4 x float> %v) {
; GFX1170-GISEL-LABEL: test_vector_reduce_fmin_v4float:
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v2, v2, v3
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -2196,10 +2018,8 @@ define float @test_vector_reduce_fmin_v4float(<4 x float> %v) {
; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_min_num_f32_e32 v2, v2, v3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -2366,15 +2186,11 @@ define float @test_vector_reduce_fmin_v8float(<8 x float> %v) {
; GFX1170-GISEL-LABEL: test_vector_reduce_fmin_v8float:
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v7, v7, v7
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v6, v6, v6 :: v_dual_max_num_f32 v1, v1, v1
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1170-GISEL-NEXT: v_dual_min_num_f32 v2, v2, v3 :: v_dual_max_num_f32 v3, v4, v4
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v4, v5, v5 :: v_dual_min_num_f32 v5, v6, v7
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v2, v2, v3
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
-; GFX1170-GISEL-NEXT: v_min3_num_f32 v1, v3, v4, v5
+; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v3, v6, v7
+; GFX1170-GISEL-NEXT: v_min3_num_f32 v1, v4, v5, v3
; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v0, v0, v1
; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -2401,15 +2217,11 @@ define float @test_vector_reduce_fmin_v8float(<8 x float> %v) {
; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v7, v7, v7
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v6, v6, v6 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX12-GISEL-NEXT: v_dual_min_num_f32 v2, v2, v3 :: v_dual_max_num_f32 v3, v4, v4
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v4, v5, v5 :: v_dual_min_num_f32 v5, v6, v7
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-GISEL-NEXT: v_min_num_f32_e32 v2, v2, v3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
-; GFX12-GISEL-NEXT: v_min3_num_f32 v1, v3, v4, v5
+; GFX12-GISEL-NEXT: v_min_num_f32_e32 v3, v6, v7
+; GFX12-GISEL-NEXT: v_min3_num_f32 v1, v4, v5, v3
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_min_num_f32_e32 v0, v0, v1
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -2667,25 +2479,17 @@ define float @test_vector_reduce_fmin_v16float(<16 x float> %v) {
; GFX1170-GISEL-LABEL: test_vector_reduce_fmin_v16float:
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v8, v8, v8
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_4)
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v9, v9, v9 :: v_dual_min_num_f32 v2, v2, v3
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v3, v6, v6 :: v_dual_max_num_f32 v6, v7, v7
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v7, v10, v10 :: v_dual_max_num_f32 v10, v11, v11
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v11, v14, v14 :: v_dual_max_num_f32 v14, v15, v15
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v5, v5, v5 :: v_dual_max_num_f32 v4, v4, v4
-; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v3, v3, v6
+; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v10, v10, v11
+; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v11, v14, v15
+; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v2, v2, v3
+; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v3, v6, v7
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1170-GISEL-NEXT: v_min3_num_f32 v6, v8, v9, v10
+; GFX1170-GISEL-NEXT: v_min3_num_f32 v7, v12, v13, v11
; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX1170-GISEL-NEXT: v_dual_min_num_f32 v7, v7, v10 :: v_dual_max_num_f32 v10, v12, v12
-; GFX1170-GISEL-NEXT: v_dual_max_num_f32 v12, v13, v13 :: v_dual_min_num_f32 v11, v11, v14
; GFX1170-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
-; GFX1170-GISEL-NEXT: v_min3_num_f32 v6, v8, v9, v7
; GFX1170-GISEL-NEXT: v_min3_num_f32 v1, v4, v5, v3
-; GFX1170-GISEL-NEXT: v_min3_num_f32 v7, v10, v12, v11
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v2, v6, v7
; GFX1170-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -2718,26 +2522,18 @@ define float @test_vector_reduce_fmin_v16float(<16 x float> %v) {
; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v2, v2, v2 :: v_dual_max_num_f32 v3, v3, v3
-; GFX12-GISEL-NEXT: v_max_num_f32_e32 v8, v8, v8
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_4)
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v9, v9, v9 :: v_dual_min_num_f32 v2, v2, v3
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v3, v6, v6 :: v_dual_max_num_f32 v6, v7, v7
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v7, v10, v10 :: v_dual_max_num_f32 v10, v11, v11
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v11, v14, v14 :: v_dual_max_num_f32 v14, v15, v15
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v5, v5, v5 :: v_dual_max_num_f32 v4, v4, v4
-; GFX12-GISEL-NEXT: v_min_num_f32_e32 v3, v3, v6
+; GFX12-GISEL-NEXT: v_min_num_f32_e32 v10, v10, v11
+; GFX12-GISEL-NEXT: v_min_num_f32_e32 v11, v14, v15
+; GFX12-GISEL-NEXT: v_min_num_f32_e32 v2, v2, v3
+; GFX12-GISEL-NEXT: v_min_num_f32_e32 v3, v6, v7
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX12-GISEL-NEXT: v_dual_min_num_f32 v7, v7, v10 :: v_dual_max_num_f32 v10, v12, v12
-; GFX12-GISEL-NEXT: v_dual_max_num_f32 v12, v13, v13 :: v_dual_min_num_f32 v11, v11, v14
+; GFX12-GISEL-NEXT: v_min3_num_f32 v6, v8, v9, v10
+; GFX12-GISEL-NEXT: v_min3_num_f32 v7, v12, v13, v11
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
-; GFX12-GISEL-NEXT: v_min3_num_f32 v6, v8, v9, v7
-; GFX12-GISEL-NEXT: v_min3_num_f32 v1, v4, v5, v3
-; GFX12-GISEL-NEXT: v_min3_num_f32 v7, v10, v12, v11
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_min_num_f32_e32 v2, v6, v7
+; GFX12-GISEL-NEXT: v_min3_num_f32 v1, v4, v5, v3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -2828,43 +2624,21 @@ define double @test_vector_reduce_fmin_v2double(<2 x double> %v) {
; GFX11-GISEL-NEXT: v_min_f64 v[0:1], v[0:1], v[2:3]
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX1170-SDAG-LABEL: test_vector_reduce_fmin_v2double:
-; GFX1170-SDAG: ; %bb.0: ; %entry
-; GFX1170-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-SDAG-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
-; GFX1170-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX1170-GISEL-LABEL: test_vector_reduce_fmin_v2double:
-; GFX1170-GISEL: ; %bb.0: ; %entry
-; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[0:1]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[2:3], v[2:3]
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1170-GISEL-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
-; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-SDAG-LABEL: test_vector_reduce_fmin_v2double:
-; GFX12-SDAG: ; %bb.0: ; %entry
-; GFX12-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-SDAG-NEXT: s_wait_expcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_samplecnt 0x0
-; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
-; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-GISEL-LABEL: test_vector_reduce_fmin_v2double:
-; GFX12-GISEL: ; %bb.0: ; %entry
-; GFX12-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-GISEL-NEXT: s_wait_expcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
-; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
-; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
+; GFX1170-LABEL: test_vector_reduce_fmin_v2double:
+; GFX1170: ; %bb.0: ; %entry
+; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1170-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
+; GFX1170-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX12-LABEL: test_vector_reduce_fmin_v2double:
+; GFX12: ; %bb.0: ; %entry
+; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT: s_wait_expcnt 0x0
+; GFX12-NEXT: s_wait_samplecnt 0x0
+; GFX12-NEXT: s_wait_bvhcnt 0x0
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
+; GFX12-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call double @llvm.vector.reduce.fmin.v2double(<2 x double> %v)
ret double %res
@@ -2973,51 +2747,25 @@ define double @test_vector_reduce_fmin_v3double(<3 x double> %v) {
; GFX11-GISEL-NEXT: v_min_f64 v[0:1], v[0:1], v[4:5]
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX1170-SDAG-LABEL: test_vector_reduce_fmin_v3double:
-; GFX1170-SDAG: ; %bb.0: ; %entry
-; GFX1170-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-SDAG-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
-; GFX1170-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1170-SDAG-NEXT: v_min_num_f64 v[0:1], v[0:1], v[4:5]
-; GFX1170-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX1170-GISEL-LABEL: test_vector_reduce_fmin_v3double:
-; GFX1170-GISEL: ; %bb.0: ; %entry
-; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[0:1]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[2:3], v[2:3]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[4:5], v[4:5], v[4:5]
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1170-GISEL-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
-; GFX1170-GISEL-NEXT: v_min_num_f64 v[0:1], v[0:1], v[4:5]
-; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-SDAG-LABEL: test_vector_reduce_fmin_v3double:
-; GFX12-SDAG: ; %bb.0: ; %entry
-; GFX12-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-SDAG-NEXT: s_wait_expcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_samplecnt 0x0
-; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
-; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX12-SDAG-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-SDAG-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[4:5]
-; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX12-GISEL-LABEL: test_vector_reduce_fmin_v3double:
-; GFX12-GISEL: ; %bb.0: ; %entry
-; GFX12-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-GISEL-NEXT: s_wait_expcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
-; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
-; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[4:5]
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
-; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[4:5]
-; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
+; GFX1170-LABEL: test_vector_reduce_fmin_v3double:
+; GFX1170: ; %bb.0: ; %entry
+; GFX1170-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1170-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
+; GFX1170-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1170-NEXT: v_min_num_f64 v[0:1], v[0:1], v[4:5]
+; GFX1170-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX12-LABEL: test_vector_reduce_fmin_v3double:
+; GFX12: ; %bb.0: ; %entry
+; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-NEXT: s_wait_expcnt 0x0
+; GFX12-NEXT: s_wait_samplecnt 0x0
+; GFX12-NEXT: s_wait_bvhcnt 0x0
+; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[4:5]
+; GFX12-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call double @llvm.vector.reduce.fmin.v3double(<3 x double> %v)
ret double %res
@@ -3160,11 +2908,6 @@ define double @test_vector_reduce_fmin_v4double(<4 x double> %v) {
; GFX1170-GISEL-LABEL: test_vector_reduce_fmin_v4double:
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[0:1]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[2:3], v[2:3]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[4:5], v[4:5], v[4:5]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[6:7], v[6:7], v[6:7]
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1170-GISEL-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
; GFX1170-GISEL-NEXT: v_min_num_f64 v[2:3], v[4:5], v[6:7]
; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -3191,11 +2934,6 @@ define double @test_vector_reduce_fmin_v4double(<4 x double> %v) {
; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[4:5]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[6:7], v[6:7], v[6:7]
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[2:3], v[4:5], v[6:7]
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -3431,22 +3169,14 @@ define double @test_vector_reduce_fmin_v8double(<8 x double> %v) {
; GFX1170-GISEL-LABEL: test_vector_reduce_fmin_v8double:
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[0:1]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[2:3], v[2:3]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[4:5], v[4:5], v[4:5]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[6:7], v[6:7], v[6:7]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[8:9], v[8:9], v[8:9]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[10:11], v[10:11], v[10:11]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[12:13], v[12:13], v[12:13]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[14:15], v[14:15], v[14:15]
; GFX1170-GISEL-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
; GFX1170-GISEL-NEXT: v_min_num_f64 v[2:3], v[4:5], v[6:7]
; GFX1170-GISEL-NEXT: v_min_num_f64 v[4:5], v[8:9], v[10:11]
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1170-GISEL-NEXT: v_min_num_f64 v[6:7], v[12:13], v[14:15]
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1170-GISEL-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_min_num_f64 v[2:3], v[4:5], v[6:7]
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -3476,22 +3206,14 @@ define double @test_vector_reduce_fmin_v8double(<8 x double> %v) {
; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[4:5]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[6:7], v[6:7], v[6:7]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[8:9], v[8:9], v[8:9]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[10:11], v[10:11], v[10:11]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[12:13], v[12:13], v[12:13]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[14:15], v[14:15], v[14:15]
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[2:3], v[4:5], v[6:7]
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[4:5], v[8:9], v[10:11]
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[6:7], v[12:13], v[14:15]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[2:3], v[4:5], v[6:7]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -3923,21 +3645,6 @@ define double @test_vector_reduce_fmin_v16double(<16 x double> %v) {
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1170-GISEL-NEXT: scratch_load_b32 v31, off, s32
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[0:1], v[0:1], v[0:1]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[2:3], v[2:3], v[2:3]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[4:5], v[4:5], v[4:5]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[6:7], v[6:7], v[6:7]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[8:9], v[8:9], v[8:9]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[10:11], v[10:11], v[10:11]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[12:13], v[12:13], v[12:13]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[14:15], v[14:15], v[14:15]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[16:17], v[16:17], v[16:17]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[18:19], v[18:19], v[18:19]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[20:21], v[20:21], v[20:21]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[22:23], v[22:23], v[22:23]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[24:25], v[24:25], v[24:25]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[26:27], v[26:27], v[26:27]
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[28:29], v[28:29], v[28:29]
; GFX1170-GISEL-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
; GFX1170-GISEL-NEXT: v_min_num_f64 v[2:3], v[4:5], v[6:7]
; GFX1170-GISEL-NEXT: v_min_num_f64 v[4:5], v[8:9], v[10:11]
@@ -3951,12 +3658,11 @@ define double @test_vector_reduce_fmin_v16double(<16 x double> %v) {
; GFX1170-GISEL-NEXT: v_min_num_f64 v[4:5], v[8:9], v[10:11]
; GFX1170-GISEL-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0)
-; GFX1170-GISEL-NEXT: v_max_num_f64 v[30:31], v[30:31], v[30:31]
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_min_num_f64 v[14:15], v[28:29], v[30:31]
-; GFX1170-GISEL-NEXT: v_min_num_f64 v[6:7], v[12:13], v[14:15]
; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1170-GISEL-NEXT: v_min_num_f64 v[6:7], v[12:13], v[14:15]
; GFX1170-GISEL-NEXT: v_min_num_f64 v[2:3], v[4:5], v[6:7]
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_min_num_f64 v[0:1], v[0:1], v[2:3]
; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -4001,21 +3707,6 @@ define double @test_vector_reduce_fmin_v16double(<16 x double> %v) {
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-NEXT: scratch_load_b32 v31, off, s32
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[2:3], v[2:3], v[2:3]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[4:5]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[6:7], v[6:7], v[6:7]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[8:9], v[8:9], v[8:9]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[10:11], v[10:11], v[10:11]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[12:13], v[12:13], v[12:13]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[14:15], v[14:15], v[14:15]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[16:17], v[16:17], v[16:17]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[18:19], v[18:19], v[18:19]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[20:21], v[20:21], v[20:21]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[22:23], v[22:23], v[22:23]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[24:25], v[24:25], v[24:25]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[26:27], v[26:27], v[26:27]
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[28:29], v[28:29], v[28:29]
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[2:3], v[4:5], v[6:7]
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[4:5], v[8:9], v[10:11]
@@ -4029,12 +3720,11 @@ define double @test_vector_reduce_fmin_v16double(<16 x double> %v) {
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[4:5], v[8:9], v[10:11]
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_loadcnt 0x0
-; GFX12-GISEL-NEXT: v_max_num_f64_e32 v[30:31], v[30:31], v[30:31]
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[14:15], v[28:29], v[30:31]
-; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[6:7], v[12:13], v[14:15]
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[6:7], v[12:13], v[14:15]
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[2:3], v[4:5], v[6:7]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_min_num_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -4060,7 +3750,5 @@ declare double @llvm.vector.reduce.fmin.v16double(<16 x double>)
;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
; GFX10: {{.*}}
; GFX11: {{.*}}
-; GFX1170: {{.*}}
-; GFX12: {{.*}}
; GFX8: {{.*}}
; GFX9: {{.*}}
>From 5307d49526292da845f934160fbe9752abd4f2ad Mon Sep 17 00:00:00 2001
From: shore <shorshen at amd.com>
Date: Thu, 17 Sep 2026 08:55:41 +0800
Subject: [PATCH 2/2] fix test
---
llvm/test/CodeGen/AMDGPU/vector-reduce-fmax.ll | 5 ++---
llvm/test/CodeGen/AMDGPU/vector-reduce-fmin.ll | 5 ++---
2 files changed, 4 insertions(+), 6 deletions(-)
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-fmax.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-fmax.ll
index ff2938f315dee..1f5d5fd71b974 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-fmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-fmax.ll
@@ -2154,9 +2154,9 @@ define float @test_vector_reduce_fmax_v8float(<8 x float> %v) {
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v2, v2, v3
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1170-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v3, v6, v7
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
; GFX1170-GISEL-NEXT: v_max3_num_f32 v1, v4, v5, v3
; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v0, v0, v1
@@ -2460,7 +2460,6 @@ define float @test_vector_reduce_fmax_v16float(<16 x float> %v) {
; GFX1170-GISEL-NEXT: v_max3_num_f32 v1, v4, v5, v3
; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_max_num_f32_e32 v2, v6, v7
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_max3_num_f32 v0, v0, v1, v2
; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-fmin.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-fmin.ll
index e4462e50850e6..57120367361f2 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-fmin.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-fmin.ll
@@ -2154,9 +2154,9 @@ define float @test_vector_reduce_fmin_v8float(<8 x float> %v) {
; GFX1170-GISEL: ; %bb.0: ; %entry
; GFX1170-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v2, v2, v3
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1170-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v3, v6, v7
+; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
; GFX1170-GISEL-NEXT: v_min3_num_f32 v1, v4, v5, v3
; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v0, v0, v1
@@ -2460,7 +2460,6 @@ define float @test_vector_reduce_fmin_v16float(<16 x float> %v) {
; GFX1170-GISEL-NEXT: v_min3_num_f32 v1, v4, v5, v3
; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_min_num_f32_e32 v2, v6, v7
-; GFX1170-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1170-GISEL-NEXT: v_min3_num_f32 v0, v0, v1, v2
; GFX1170-GISEL-NEXT: s_setpc_b64 s[30:31]
;
More information about the llvm-commits
mailing list